diff --git a/.testcoverage.yml b/.testcoverage.yml index 451014a32..fe2c658f4 100644 --- a/.testcoverage.yml +++ b/.testcoverage.yml @@ -79,6 +79,17 @@ exclude: - ^internal/settings/ - ^cmd/wavehouse/validate\.go$ - ^cmd/wavehouse/bootstrap\.go$ + # The DynamoDB dedupe backend: the e2e binary runs Pebble dedupe, so + # this file measured 0% there and pulled e2e to 58.6%. The unit + # (fake API) and integration (dynamodb-local) suites cover it, and the + # merged total still counts it. + - ^internal/dedupe/dynamodb\.go$ + # wireDynamoDedupe and its retry component (internal/app/wire_dynamodb.go): + # same reason as dynamodb.go above — the e2e binary never selects + # dedupe.backend: dynamodb, so this file measured 0% there and pulled + # e2e to 59.7%. The unit and integration suites cover it, and the + # merged total still counts it. + - ^internal/app/wire_dynamodb\.go$ # The in-process cache backend: the e2e stack runs cache.backend=redis # (#613), so the binary carries LocalCache and its version index but e2e # never reaches them. The unit suite and the integration suite's main diff --git a/AGENTS.md b/AGENTS.md index 3022e1b7b..4573713a9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -34,13 +34,13 @@ Twenty internal packages under `internal/` (plus `internal/testutil/` for shared - **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index), and `RedisCache`, the Redis-compatible shared backend (random version tokens under the tenant's hash tag, one-round-trip lookups, bypass on failure behind a circuit breaker, deferred invalidations retried; selected by `cache.backend: redis`, configured by the boot config's `cache.redis` block — [#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every key carries the tenant (in `RedisCache`, after the key prefix: `:{}:…` for a version token, `:q::…` for a value); in `LocalCache` and the version index it leads ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for the caller's query key and its singleflight, escaped whole as the lead field of the stored key `|.||…`, where each raw table and scope name is escaped by `keyenc` (a `Namespace` carries them raw, so no caller escapes); the index holds a version per tenant, per (tenant, table) and per (tenant, table, scope), keyed by raw name and bumped in place (one entry per live namespace however often it is bumped, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)) — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` drops the tenant's index so its next key gets a process-unique generation, orphaning its every cached result in one step, pipe results included (no insert reaches a pipe result until [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); `Lookup` returns a `Snapshot` of the versions it read, taken before the handler chooses any input a bump invalidates — the tenant's connection included — and `Set` files the fill under it, so a write landing mid-query, or a reload moving the tenant to another address or database after the request took its connection, orphans the fill ([#382](https://github.com/Wave-RF/WaveHouse/issues/382)), and every backend runs the conformance suite `internal/testutil/cachetest`; the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the whole cache of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) - **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (nil for a tenant on no pool, a `503`), `Target` (the tenant's own HTTP wiring over its pool's TLS config), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config. `Classify` (`errclass.go`) says what a failed ClickHouse request means for the request — `Unavailable`, `Denied`, `Rejected` (any unlisted exception code: the server read it and refused it), or `Unknown` (no code, no recognizable transport failure) — over the driver's error types and the HTTP interface's `HTTPError`; the ingest worker and the query handlers (`api/ch_errors.go` `writeCHError`, [#403](https://github.com/Wave-RF/WaveHouse/issues/403), [#271](https://github.com/Wave-RF/WaveHouse/issues/271)) both use it - **`chsql/`** — dependency-free ClickHouse SQL helpers shared by `query`/`policy` (avoids an import cycle): `QuoteIdent` (backtick-quote every identifier) + `BindUnsafe` (reject names with a literal `?`) -- **`config/`** — YAML + env var config loading (cleanenv); strict on both sides (undeclared YAML key, unbound `WH_*` variable) and probes `data_dir` writability when a selected backend keeps state there (`NeedsDataDir`); `backends.go` holds each layer's `.backend` (the in-process value by default; `cache.backend` also takes `redis`, whose sub-block is `cache_redis.go`); `config.go` holds `roles` (`Has(Role)`) and `instance_id`, and `Validate` refuses a role split the backends cannot serve (any split over the embedded MQ; `api` without `ingest`, or the reverse, over a local cache) — boot is the validator, there is no dry run +- **`config/`** — YAML + env var config loading (cleanenv); strict on both sides (undeclared YAML key, unbound `WH_*` variable) and probes `data_dir` writability when a selected backend keeps state there (`NeedsDataDir`); `backends.go` holds each layer's `.backend` (the in-process value by default; `cache.backend` also takes `redis`, whose sub-block is `cache_redis.go`, and `dedupe.backend` takes `dynamodb`, with its `dedupe.dynamodb` sub-block); `config.go` holds `roles` (`Has(Role)`) and `instance_id`, and `Validate` refuses a role split the backends cannot serve (any split over the embedded MQ; `api` without `ingest`, or the reverse, over a local cache) — boot is the validator, there is no dry run - **`coord/`** — leases for work that must run in one process at a time: `Coordinator.TryAcquire(ctx, name)` → a `Term` (fencing `Token`, strictly increasing per name; `Done`/`Err`, `ErrLost` on loss; `Resign`), `ErrHeld` while another holder's — or this coordinator's own — term is live; `RunElected` runs a loop only while holding its lease, resigning when the loop returns and campaigning again every `RetryPeriod`. `Local` is the in-process implementation (first taker wins, never expires; `Peer` is a second handle over the same table for tests); every implementation runs `coordtest.Conformance`. Imports only the standard library, so a distributed backend lives beside its connection (NATS KV in `internal/mq`). `internal/app`'s `wireCoord` opens the one `coord.backend` selects and the sweeper runs through `RunElected` under the `sweeper` lease -- **`dedupe/`** — `Deduplicator` interface → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) +- **`dedupe/`** — `Deduplicator` interface (two-phase `Reserve`/`Commit`/`Release` over `Key{Table, ID}`; every backend passes the `dedupetest` conformance suite) → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant and table, pending claims in memory, committed ids stored with their expiry and deleted by an hourly background sweep along with the version-0 keys from before the table joined the key, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges) or `Dynamo` (one shared DynamoDB table, conditional `PutItem` claims; conformance-tested against dynamodb-local, selected by `dedupe.backend: dynamodb`; boot checks the table and never creates it outside dynamodb-local), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` or, gated on the table check (`Factory.Gated`), `Dynamo.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) - **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version, + `Validate()` for ingest payloads + `CanonicalizeTimestamps()` rewriting top-level `DateTime`/`DateTime64` column values to the canonical RFC 3339 UTC wire form pre-publish (Key Design Decision #19) - **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else — `row` is one positional `JSONCompactEachRow` line and `columns` names its slots, the table's **insertable** columns (a `MATERIALIZED`/`ALIAS` column cannot be named in an `INSERT`); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs. In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) -- **`keyenc/`** — the one escaping composite keys are built from: `Escape` keeps `[A-Za-z0-9_-]` (exactly the tenant-id grammar, so a tenant id is its own escaped form) and writes every other byte as `%XX`, `Unescape` is `url.PathUnescape` (lenient: either hex case, and a byte left unescaped reads as itself, so a `%2D` an earlier build wrote still reads), `Join`/`AppendJoin` escape each field and put a separator between them (they panic on no fields, and on a separator the escaping could write or one outside ASCII) and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to and no field reaches a key unescaped: NATS subjects (`Join`/`Split` after the verbatim tenant) and the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — use it; changing what it keeps orphans every stored key, and on the shared backend, whose keys every process builds for itself, splits them between builds for the length of a rolling upgrade (a bump one build makes misses the entries the other filed, served until their TTL) -- **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20). and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal, `ErrUnavailable` a broker that cannot be reached — both a `503`, with `Retry-After` `30` and `5`), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, stream names, sequences, and ack floors are private to the one implementation, `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`, `deadletter.go`), whose subject tokens are escaped by the shared `internal/keyenc`; `internal/app` constructs it and hands everything else a `mq.Broker`. Every implementation passes the conformance suite in `internal/mq/mqtest` (`mqtest.Run`), which states the `Broker` contract as behavior; a new backend runs it from its own test, with `mqtest.Caps` only where its semantics legitimately differ +- **`keyenc/`** — the one escaping composite keys are built from: `Escape` keeps `[A-Za-z0-9_-]` (exactly the tenant-id grammar, so a tenant id is its own escaped form) and writes every other byte as `%XX`, `Unescape` is `url.PathUnescape` (lenient: either hex case, and a byte left unescaped reads as itself, so a `%2D` an earlier build wrote still reads), `Join`/`AppendJoin` escape each field and put a separator between them (they panic on no fields, and on a separator the escaping could write or one outside ASCII) and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to and no field reaches a key unescaped: NATS subjects (`Join`/`Split` after the verbatim tenant) and the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — and the dedupe keys (`/
/`) use it; changing what it keeps orphans every stored key (an orphaned dedupe key lets a seen id through again), and on the shared backend, whose keys every process builds for itself, splits them between builds for the length of a rolling upgrade (a bump one build makes misses the entries the other filed, served until their TTL) +- **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20). and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal, `ErrUnavailable` a broker that cannot be reached — both a `503`, with `Retry-After` `30` and `5`; `WithIdempotencyKey` makes a republish inside the queue's duplicate window a no-op), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, stream names, sequences, and ack floors are private to the one implementation, `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`, `deadletter.go`), whose subject tokens are escaped by the shared `internal/keyenc`; `internal/app` constructs it and hands everything else a `mq.Broker`. Every implementation passes the conformance suite in `internal/mq/mqtest` (`mqtest.Run`), which states the `Broker` contract as behavior; a new backend runs it from its own test, with `mqtest.Caps` only where its semantics legitimately differ - **`observability/`** — OpenTelemetry pipeline: `InitProvider` wires trace/metric/log providers via OTLP gRPC (each signal independently gated). A top-level `Prometheus` config block drives an optional `/metrics` scrape endpoint that runs independently of OTLP push — standalone (Alloy/Mimir scrape, no collector), alongside OTLP, or off. `NewLogger` produces a slog handler that fans out to stdout AND OTLP (stdout always 100%, OTLP sample-rate-aware). `TraceHandler` injects trace_id/span_id from active spans. `tracer.go` provides W3C trace context propagation over message headers (`InjectHeaders`/`ExtractHeaders` on a plain header map; `internal/mq` injects on every publish and extracts on the `Subscribe` path, so this package never sees a NATS type). - **`pipes/`** — Named query pipes: `NamedQuery` type + `BindParams` + `Source` (read per request; `settings.Store` in production, `Static(q...)` in tests) - **`policy/`** — Hasura-style access control, **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant splits by operation into `SelectPermissions` (columns, row `filter`, aggregations, the `max_*` limits) and `InsertPermissions` (columns, `check`) — so a field only one side honors does not exist on the other. `Evaluate()` resolves ONE operation and leaves the other side **nil** (`Select *ResolvedSelect` / `Insert *ResolvedInsert`), which every accessor fails closed on — nil is "not resolved", distinct from an empty side, which is "unrestricted" (what the admin return builds). Claim templating (`{{ jwt.claim.path }}`) resolves during that call. Policies come from `Source`, a `func() *Policy` read per call (`settings.Store.Policy` in production, `Static(p)` in tests) @@ -60,7 +60,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 5. **Per-tenant-table batching** — the worker groups events by tenant table (the tenant read off each message's `mq.Topic`), so one INSERT never mixes tenants and a batch invalidates its own tenant's cache namespaces; then it splits each batch by column list (`groupByColumns`), emitting one `INSERT INTO … (cols) FORMAT JSONCompactEachRow` per distinct list so a schema change mid-stream can't corrupt a statement. Each tenant table's batch is independent. 6. **Dead Letter Queue** — batch inserts ClickHouse **rejects** (isolated row by row; `chconn.Classify` == `Rejected` — a multi-row batch refused for its size, `chconn.Splittable`, is split row by row too) publish to the tenant's own dead-letter queue (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). A ClickHouse that cannot take the insert — unavailable, denied, or no verdict — never dead-letters a row, not even mid-isolation: the rows go back to the MQ with a delayed nak under a per-pool backoff — per table for a failure of one table (`chconn.TableScoped`: read-only, too many parts or mutations, a missing grant; `internal/ingest/backoff.go`), counted by `wavehouse_ingest_retries_total`. No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format`, or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. 7. **Auth: always on, fail-loud, decoupled from authz (security)** — the JWT middleware always runs (no `auth.enabled`/`dev_mode` flag); it verifies with HMAC **or** JWKS (not both), with accepted `alg` pinned to the active verifier and checked before any key is used (rejects `alg:none` and cross-family confusion). No/invalid/expired token → empty role → policy `default_role`, with the bad-token reason stashed so a denying gate returns a loud `401`, not a bare `403`; the one token outcome that never reaches `default_role` is a verifier still fetching its JWKS (`auth.ErrVerifierPending` → `503` + `Retry-After`, `api.refuseUnverifiable`). Elevated access needs a valid granted role. **Sanctioned exception:** a configured non-JWT operator key (`auth.operator_key`; presented via `Authorization: Operator ` or the `X-Operator-Key` alias) deliberately couples authN+authZ — a constant-time match authorizes a full-access platform operator (stamps the admin role plus an operator bit) independent of the verifier (see #11). Detail: architecture.md § `api/` + `internal/auth`; see also #11, §Security Considerations. -8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key, overridable per table. +8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant and table; claims are two-phase, one call per phase per window of up to 256 records — `Reserve` → publish (under the id's idempotency key) → `Commit`, or `Release` when the publish definitely failed, while one whose outcome is unknown is left to lapse; a store that cannot answer is a `503` + `Retry-After`; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key and `dedupe.retention` how long a committed id stays a duplicate (`"0"` = forever, else at least the queue's two-minute duplicate window), both overridable per table. 9. **Singleflight** — the cached read handlers coalesce concurrent misses (`x/sync/singleflight`) under the tenant-led cache key to prevent cache stampede, per tenant. 10. **Active Sweeper** — purges NATS messages that are both ACKed (written to CH) and older than the gap window; SSE gap-fill uses `DeliverByStartTime`, no in-process ring buffer. It runs only in the process holding the `sweeper` lease (`coord.RunElected`); that lease is not fenced: an overlap cannot lose ClickHouse data, since every sweep stops at the consumer's ack floor; it can only trim SSE replay history, and only when the two holders' settings views differ (one still reading a shorter `stream.gap_window_minutes`, or missing a tenant, after a reload the other has applied) — which fencing would not prevent either. Anything that does need exclusivity must check the term's `Token`. 11. **Hasura-style access control: fail-closed (security)** — `policy.IsAdmin` (role == `admin_role`, **exact case-sensitive**, default `"admin"`) is the single admin check, shared by `Evaluate`/`ResolveRole`/`Validate`/the `/v1/ops` gate/`RoleAllowed`. Empty/absent role matches nothing (no `"*"` wildcard); `Validate` rejects empty role keys; a `nil` policy (deleted) denies **everyone incl. admin** via a role — a total lockout for token-based callers, so recovery is writing `policies.json` and reloading, never an implicit admin grant (**exception:** the operator key's `auth.IsOperator` bit passes the `/v1/ops` gate even under a `nil` policy — a deliberate break-glass that can `POST /v1/ops/settings/reload` over HTTP, see #7). Over a nested settings directory the `/v1/ops` gate reads no policy at all — those routes reach every tenant, so the operator key alone passes and an admin-role token gets `403`; `api.NewRouter` decides that from the registry's shape, not from what was wired. `default_role` is the one sanctioned roleless exception (`ResolveRole` maps empty → it pre-eval); `default_role == admin_role` is permitted but dev-only and loudly warned (`policy.DefaultRoleGrantsAdmin`). Preserve when touching `internal/policy` (policy twin of #13; see #159). Detail: architecture.md § `policy/`. @@ -153,7 +153,7 @@ If `make ci` passes locally, your commit has crossed the same gates CI will run ### Running `make ci` (for agents) -`make ci` is **self-contained**: the integration suite (`tests/integration/`) and the E2E orchestrator (`scripts/orchestrator/`) each boot ClickHouse and a Redis via **testcontainers on random host ports**, and the shared cache backend's integration tests (`internal/cache/`) start their own Redis, Valkey, Dragonfly and one-node Redis Cluster containers the same way. The only prerequisite is a running **Docker daemon** — do **not** `make deps-up` or start ClickHouse first (`deps-up` is for `make dev` only). +`make ci` is **self-contained**: the integration suite (`tests/integration/`) and the E2E orchestrator (`scripts/orchestrator/`) each boot ClickHouse and a Redis via **testcontainers on random host ports** (the integration suite also dynamodb-local), and the shared cache backend's integration tests (`internal/cache/`) start their own Redis, Valkey, Dragonfly and one-node Redis Cluster containers the same way. The only prerequisite is a running **Docker daemon** — do **not** `make deps-up` or start ClickHouse first (`deps-up` is for `make dev` only). Run it via the **background Bash tool** (`run_in_background: true`) and wait for the completion notification; the harness re-invokes you on exit, so polling the log with `tail` only burns context: @@ -436,10 +436,10 @@ internal/chconn/ → ClickHouse pools, one per connection tuple among the internal/chsql/ → Shared ClickHouse SQL helpers (identifier quoting + bind-safety) internal/config/ → Configuration structs + loader internal/coord/ → Leases with fencing tokens (interface, in-process Local, RunElected, coordtest conformance suite) -internal/dedupe/ → Optional deduplication (interface + embedded/distributed) +internal/dedupe/ → Optional deduplication (Reserve/Commit/Release interface; Pebble, DynamoDB) internal/discovery/ → ClickHouse schema introspection + ingest validation internal/ingest/ → Batch buffer with DLQ + Active Sweeper (NATS message lifecycle) -internal/keyenc/ → One escaping for composite keys (NATS subject tokens, cache keys) +internal/keyenc/ → One escaping for composite keys (NATS subject tokens, cache keys, dedupe keys) internal/mq/ → MQ boundary (the only NATS/JetStream importer: owned message/consumer/stream types + embedded server; mqtest/ is the Broker conformance suite) internal/observability/ → OpenTelemetry pipeline (traces/metrics/logs providers, Prometheus exporter, slog fan-out, message-header trace propagation) internal/pipes/ → Named query pipes (types, parameter binding, Source) @@ -448,7 +448,7 @@ internal/query/ → Structured query AST + SQL builder internal/settings/ → Settings directory (validate, adopted snapshot + reload, watcher, embedded seed) internal/stream/ → SSE fan-out (event Hub: project once per role, Subscriber outbound queue, Bucket fan-out, keepalive Heartbeater wheel) internal/tenant/ → Tenant id (type, grammar, reserved default, request header name) -internal/testutil/ → Shared test helpers (mocks, JWT + schema helpers; logtest/ captures or silences the default logger; cachetest/ is the conformance suite every cache.Cache backend runs; mutationtest/ holds the shared write-classifier cases) +internal/testutil/ → Shared test helpers (mocks, JWT + schema helpers; logtest/ captures or silences the default logger; cachetest/ is the conformance suite every cache.Cache backend runs; mutationtest/ holds the shared write-classifier cases; storedir/ is the embedded broker's store directory in tests, removed once late consumer-state writes land) tests/ → Integration & E2E tests tests/integration/ → Go integration tests (//go:build integration; ClickHouse testcontainer, and Redis for shared_cache_test.go). A package tested against its own external server keeps them beside it: internal/cache/redis_integration_test.go (Redis, Valkey, Dragonfly, Redis Cluster testcontainers) tests/e2e/ → E2E test stack (scripts/orchestrator boots ClickHouse and Redis testcontainers + the wavehouse-cov binary) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5c69ebc50..3a8ec686d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,13 +10,15 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Added +- **`dedupe.backend: dynamodb` selects the shared DynamoDB dedupe table** (`internal/config/{backends,config}.go` (+ tests), `internal/app/{app,wire,wire_dynamodb}.go` (+ `dedupe_dynamodb_test.go`), `internal/dedupe/{stores,dynamodb}.go` (+ tests), `tests/integration/dedupe_dynamodb_app_test.go` (new), `.testcoverage.yml`, `config.yaml`, `AGENTS.md`, `docs/src/content/docs/{configuration.mdx,settings-directory.mdx,deployment.md,architecture.md,api.md,sdk/reference.md}`): PR F5 of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). Pods that set it share seen ids, so an id ingested through one is a duplicate through every other. New boot keys: `dedupe.lease` (`WH_DEDUPE_LEASE`, `30s`, how long a claimed id stays pending and the in-flight `503`'s `Retry-After`; at most `59s` with the embedded queue, so that the lease plus its own ceiling to the next second plus one more second fits its 2-minute duplicate window: a client obeying that `Retry-After` after an uncertain publish can republish as late as the lease plus that ceiling plus a second after the claim, since DynamoDB rounds a claim's expiry up to the second), `dedupe.reserve_concurrency` (`WH_DEDUPE_RESERVE_CONCURRENCY`, `64`, which also sizes the DynamoDB client's idle connections per host; ingest sends a window of up to 256 ids per call), and the `dedupe.dynamodb` block (`table` (required), `region`, `endpoint`, `timeout` `250ms`, `max_attempts` `3`, `retry_mode` `standard`/`adaptive`, `create_table`), each with its `WH_DEDUPE_DYNAMODB_*` variable. Their defaults are in `defaults()` like every boot key's, so an explicit `0` lease, concurrency, timeout or attempt count, or an empty `retry_mode`, refuses boot rather than becoming the default (the `dynamodb` block's only while `dynamodb` is selected). Credentials come from the AWS SDK's default chain, never from config. Boot checks the table (key schema `pk` String alone; TTL off on `ex` is a warning) in a process running the `api` role, the one that opens the dedupe stores, whether or not a tenant has dedupe on: a misconfigured table (missing, the wrong key schema, access denied) refuses boot over a flat settings directory whose tenant has dedupe on and is logged at `ERROR` otherwise; any other failure (a throttle, a timeout, the network), a nested directory, or no tenant deduping yet boots and fails every switched-on tenant's ingest closed until the check passes, retried in the background (1s backing off to 30s) and at once after every reload. A reload makes no table call and does not wait on a tenant whose dedupe setting is unchanged: it holds the lock that serializes reloads, so it applies each tenant's switch against the last check's result and only wakes the retry; it waits only for a tenant whose store it closes — dedupe switched off, or the tenant removed or rejected — and then only for that tenant's in-flight calls, before its store closes. No region from the config or the SDK chain refuses boot. `create_table` creates a missing table at boot (an endpoint not up yet is a transient failure, retried like the check) and is refused unless `endpoint` is set, so it only ever reaches dynamodb-local. +- **A DynamoDB dedupe backend** (`internal/dedupe/dynamodb.go` (new, + tests), `tests/integration/{setup,dedupe_dynamodb}_test.go`, `go.mod`, `AGENTS.md`, `docs/src/content/docs/{architecture,deployment}.md`): PR F3 of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). `dedupe.Dynamo` keeps every tenant's seen ids in one shared table, so pods that share the table also share seen ids, which the per-process Pebble store cannot do. The table's only key is a String `pk` holding the readable dedupe key (`acme/clicks/evt-123`), so an item reads as-is in the console. `Reserve` is a conditional `PutItem` that is atomic across pods; an SDK retry of a put DynamoDB applied but whose answer was lost (a `500`, a reset connection) finds its own item by token and keeps the claim, rather than answer `InFlight` and hold the id for the lease. `Commit` is `BatchWriteItem`, retrying for up to eight jittered rounds both the items DynamoDB leaves unprocessed and a batch it throttled whole, and `Release` is a `DeleteItem` conditional on the claim's token. A client that disconnects mid-`Reserve` does not strand its claim: puts already sent run to their answer on a context its cancellation does not reach, and are then released, so its retry is not answered `InFlight` for the lease. Only a put cut off by its own call timeout (which DynamoDB may apply after the release), or a release that fails, still holds its id until the lease ends. An expired item counts as absent without waiting for TTL. Each call has a 250 ms timeout covering the SDK's three attempts, whose retries back off with full jitter under a ceiling that keeps their waits within half the timeout, so a throttled call fails with the throttle as its cause rather than on the deadline. Throttling, timeouts and an unreachable table wrap `ErrUnavailable`, and five such failures in a row within a second short-circuit claims for a second. Credentials come from the AWS SDK's default chain. The HTTP client keeps one idle connection per host for each of the 64 calls a `Reserve`, `Commit` or `Release` runs at once (the SDK's default keeps 10), so a warm 64-key `Reserve` reuses every connection rather than open about 50. WaveHouse never creates the production table: `CreateTable` works only against dynamodb-local, and the Deployment page carries an example Terraform table and IAM policy. The backend passes the `dedupetest` conformance suite against a pinned `amazon/dynamodb-local` container, along with 32 clients racing one id, injected throttles and an unreachable endpoint. New metrics: `wavehouse_dedupe_dynamodb_requests_total`, `_request_duration_seconds`, `_unprocessed_items_total`, `_short_circuits_total`. `dedupe.backend: dynamodb` selects it (entry above). New dependencies: `aws-sdk-go-v2` (`service/dynamodb`, `config`) and what they require. - **`cache.backend: redis` shares the query cache across instances** (`internal/config/{cache_redis,backends,config}.go` (+ tests), `internal/app/wire.go` (+ tests), `internal/ingest/worker.go`, `tests/integration/shared_cache_test.go`, `scripts/orchestrator/main.go`, `tests/e2e/fixtures/config.yaml`, `.testcoverage.yml`, `deployments/compose/dependencies.yaml`, `config.yaml`, `.github/workflows/{ci.yml,README.md}`, `docs/src/content/docs/{configuration.mdx,deployment.md,architecture.md,settings-directory.mdx,getting-started.md,pipes.mdx,api.md,development.md,index.mdx,why-wavehouse.md,sdk/reference.md}`, `README.md`, `AGENTS.md`): PR E4 of the distributed-deployment epic ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). The boot config's `cache.backend` now takes `redis`, configured by a new `cache.redis` block (`WH_CACHE_REDIS_*`): `addrs` (required), `mode` (`standalone` or `cluster`; `sentinel` refuses boot until [#656](https://github.com/Wave-RF/WaveHouse/issues/656)), `username`, `password` (a secret — set it through the environment), `db`, `tls.{enabled,ca_file,cert_file,key_file,server_name,insecure_skip_verify}`, `key_prefix` (`wh`), `timeout` (`100ms`) and `dial_timeout` (`1s`), each at most `1s` since boot and shutdown each wait out a connection attempt they bound, `max_value_bytes` (1 MiB), `compress_min_bytes` (`1024`; `0` never compresses) and `version_ttl` (`168h`). Every instance pointed at one server shares its cached results, and an insert on any instance invalidates every instance's. It is also the shared cache a split of [`roles`](https://github.com/Wave-RF/WaveHouse/pull/622) needs: the refusal of `api` without `ingest`, or the reverse, over `cache.backend: local` now names `redis`, though every split is still refused while the queue is embedded. A malformed block — no address, an address without a valid port, more than one address in `standalone` mode, a URL-style address (refused without repeating it, since it may carry a password), an unknown mode, `db` other than `0` in cluster mode, an unreadable TLS file, a TLS key set while `tls.enabled` is off — refuses boot; an unreachable server, or one that rejects the password, does not: the process boots with the cache bypassed and keeps reconnecting, logging a rejected password at `ERROR` on every attempt, so a rotated secret cannot crash-loop every instance at once. `insecure_skip_verify`, and a `cache.redis.addrs` set while `cache.backend` is `local`, are logged at `WARN` at boot. The ingest worker's log of an invalidation that did not land drops from `ERROR` to `WARN`, since the shared backend defers and retries it: an outage would otherwise log an `ERROR` for every batch. The e2e suite now runs against a Redis testcontainer with `cache.backend: redis`, so the shared backend is exercised end to end; the e2e per-suite exclude now names what its run still can't reach instead — `internal/cache/pending.go` (the retry of an invalidation the server did not take, which needs an outage), `internal/config/cache_redis.go` (the block's own rejection paths) and `internal/cache/(local|version_manager).go` (the `local` backend, which e2e no longer runs) — and the unit and integration suites keep covering them. An integration test boots two instances over one Redis and one ClickHouse: a result one fills is a hit for the other, and a row ingested through one is served fresh by the other on its next query, well inside the stale entry's TTL; another pauses Redis and checks queries keep succeeding from ClickHouse, then turn back to hits; a third runs the first one's hit, insert and fresh-miss lifecycle on the suite's own `cache.backend: local` app, since e2e no longer exercises that backend. `deployments/compose/dependencies.yaml` gains an optional `redis` profile for local multi-instance work. The deployment guide gains a "Multiple instances and the shared cache" section: what each instance keeps to itself, what a reader on another instance can see and when, `maxmemory-policy`, and the metrics to alert on. - **A Redis-compatible shared cache backend** (`internal/cache/{redis,redis_codec,breaker,pending,metrics}.go` (+ tests), `internal/cache/cache.go`, `internal/cache/redis_integration_test.go`, `Makefile`, `go.mod`, `.testcoverage.yml`, `docs/src/content/docs/{architecture,development}.md`, `AGENTS.md`, `CONTRIBUTING.md`): PR E3 of the distributed-deployment epic ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). `cache.RedisCache` keeps query results and their versions in one Redis, Valkey, Dragonfly, ElastiCache or MemoryDB server shared by every process, so an insert one process makes invalidates what every other process has cached. Versions are random tokens, one per tenant, table and scope, under the tenant's hash tag, with the table and scope escaped into the key (`internal/keyenc`) so no two names share a token; a bump sets a fresh one, and a value carries the tokens it was computed under, so a lookup is one pipelined round trip (`MGET` of the tokens plus `GET` of the value, no scripts) and a token lost to eviction, expiry, `FLUSHALL` or a restart without persistence can only cause misses — `maxmemory-policy allkeys-lru` is safe. Restoring an RDB or AOF snapshot, or a backup, is a rollback instead (a restart after a crash that reloads the server's last save included, which stock Redis and Valkey make by default): the old tokens return with their values, so invalidations made since are undone until those entries' TTL. Values of 1 KiB or more are zstd-compressed when that makes them smaller, and a value over 1 MiB stored is not cached. A server that fails or takes longer than the per-operation timeout (100 ms) is a miss, a skipped fill and a deferred invalidation, never a failed query; five failures in a row, or one reply refusing writes (`READONLY` from a demoted primary, `OOM` when full under `noeviction`, and the like) or the credentials (`WRONGPASS` or `NOAUTH` after a password rotation), open a circuit breaker — logged once per opening, at `ERROR` for the credentials and `WARN` otherwise, while a failed probe reopening it every 5 s logs only at `DEBUG` unless its cause changed — that skips the server until a probe write succeeds within the per-operation timeout (connections are replaced every minute, so after a failover behind a stable address the process reaches the new primary, and delivers the bumps it owes, within about a minute; the probe also allows up to twice the dial timeout for a reconnect, but the client's other connections must reconnect within the per-operation timeout, so that timeout should still exceed a reconnect), and deferred invalidations are retried until they land — the first at once, and at the probe's cadence while the breaker is open — collapsing to one tenant-wide bump per tenant past 100,000 keys; until one lands, the process that owes it bypasses the lookups it would orphan. New metrics: `wavehouse_cache_lookups_total{backend,result}`, `wavehouse_cache_op_duration_seconds{backend,op}`, `wavehouse_cache_breaker_open{backend}`, `wavehouse_cache_invalidations_total{backend,result}`, `wavehouse_cache_invalidations_pending{backend}`, `wavehouse_cache_value_bytes{backend}`, `wavehouse_cache_oversize_total{backend}`, `wavehouse_cache_set_failures_total{backend,reason}`. `cache.backend: redis` selects it (the entry above). Tested against Redis 8.10, Valkey 8.1, Dragonfly 2.0 and a Redis Cluster node by the conformance suite, plus lost-token, snapshot-rollback, compression, stored-size, server-stops-answering, refused-writes (a demoted primary, a full `noeviction` server), rotated-credentials, slow-reconnect (over TLS), slow-server, failover-behind-a-stable-address and owed-invalidation cases; `make test-integration` now also runs `internal/cache`'s integration-tagged tests. Adds `github.com/redis/rueidis` (Redis org, Apache-2.0; its only runtime dependency is `golang.org/x/sys`) and makes `github.com/klauspost/compress` a direct dependency. - **One conformance suite for every `mq.Broker`, and a transient broker failure is a `503`** (`internal/mq/mqtest/` (new: the suite and the embedded broker's run of it), `internal/mq/{mq,embedded}.go` (+ tests), `internal/api/ingest.go` (+ tests), `.testcoverage.yml`, `docs/src/content/docs/{api,architecture}.md`, `AGENTS.md`): the first piece of the external-NATS workstream of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). `mqtest.Run` states the `Broker` contract as behavior — round trips with names that need encoding, per-tenant order, `Nak` and `AckWait` redelivery, the trace context reaching `Subscribe`, dead-lettering that keeps the topic and leaves the original unacked, per-tenant dead-letter counts, replay bounds and isolation, and exactly one `failed` report when delivery ends underneath a consumer — through the interfaces alone, so the external backend runs the same cases, with `mqtest.Caps` for the four places where its semantics legitimately differ. The embedded broker passes it; writing it turned up that a durable deleted on several tenants' queues could report on `failed` more than once, which is fixed and pinned by a test that deletes it on one queue after another. It also turned up a replay that lost its connection mid-pull passing for a caught-up one when the pull ended in a timeout; that is an error now, as the `Replayer` contract says. The interface comments now allow a delivery unit that is a partition holding several tenants, a `CreateConsumer` that finds a durable rather than creating one, a `PurgeAcked` that leaves retention to the operator, and zero dead-letter counts where there is no per-tenant queue. A new sentinel, `mq.ErrUnavailable`, is a broker that cannot be reached or does not answer in time: the ingest handler answers it with `503` + `Retry-After: 5` rather than the `500` "publish failed" it would have been. Nothing returns it yet; the external backend of #613 will. - **Process roles: the API and the background workers can run in separate processes** (`internal/config/config.go` (+ `roles_test.go`, `defaults_test.go`), `internal/config/backends.go`, `internal/app/{app,wire}.go` (+ `roles_test.go`), `internal/api/router.go` (+ tests), `tests/integration/{setup,tenants}_test.go`, `config.yaml`, `docs/src/content/docs/{configuration.mdx,deployment.md,architecture.md}`, `AGENTS.md`), part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). The boot config gains `roles` (`WH_ROLES`, default `api,ingest,sweeper`, set in `defaults()` like every boot default, so an explicit `roles: []` refuses boot) and `instance_id` (`WH_INSTANCE_ID`, default `-<8 hex>`, fresh at every boot; logged at boot, and recorded as a lease holder once a shared `coord.backend` exists). A process wires only what its roles need: `api` runs the HTTP API with schema discovery, the token verifiers, the dedupe stores and the SSE hub (all per API process); `ingest` runs the ingest worker; `sweeper` runs the sweeper under its lease. A process without `api` serves an ops-only listener on `server.port` (the probes and their aliases, `/version`, the same-port metrics path, and `POST /v1/ops/settings/reload`, which takes the operator key alone); every other route answers 404, under `/v1/ops` once the operator key has passed. Boot refuses any split over the embedded MQ, which no other process can reach, and `api` without `ingest` (or the reverse) over a local cache, which the ingest worker's invalidations would never reach. Until a shared `mq.backend` exists, every process therefore runs every role, which is the default, so nothing changes for an existing deployment. `data_dir` is probed for Pebble only in a process running `api`. A `config.Config` built without `config.Load` must now name its roles (`config.AllRoles()` for all of them): `app.New` refuses an empty set. - **Leases for work that must run in one process at a time, and the sweeper runs under one** (`internal/coord/` (new: `coord.go`, `local.go`, `elect.go`, `coordtest/`, + tests), `internal/app/{app,wire}.go` (+ tests), `internal/config/backends.go`, `internal/ingest/sweeper.go`, `.github/labeler.yml`, `.testcoverage.yml`, `docs/src/content/docs/{architecture,development,ingest-pipeline}.md`, `docs/src/content/docs/configuration.mdx`, `config.yaml`, `AGENTS.md`): part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). `coord.Coordinator` hands out named leases (`TryAcquire` → a `Term` with a strictly increasing fencing `Token`, a `Done` channel and `Resign`; `ErrHeld` while another holder's term is live), and `coord.RunElected` runs a loop only while its process holds the lease, resigning when the loop returns and campaigning again every 2s. `coord.Local` is the in-process implementation, and `coordtest.Conformance` is the suite every implementation runs — the NATS KV backend that lets several replicas share one queue comes next. The sweeper now runs through `RunElected` under the `sweeper` lease; with the in-process coordinator the one process always holds it, so nothing changes for a single-process deployment beyond one `coord: elected` log line at startup. `coord.backend` now selects the coordinator (`local`, the only value), so a `config.Config` built without `config.Load` must name it as well as the other three layers' backends. -- **Each layer's implementation is chosen at boot** (`internal/config/{backends,config}.go` (+ tests), `internal/config/defaults_test.go`, `internal/app/{app,wire}.go` (+ tests), `cmd/wavehouse/main.go`, `tests/integration/{setup,tenants}_test.go`, `config.yaml`, `docs/src/content/docs/{configuration.mdx,settings-directory.mdx,architecture.md}`): the first step of running WaveHouse as more than one process ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). The boot config gains `mq.backend` (`WH_MQ_BACKEND`, default `embedded`), `cache.backend` (`WH_CACHE_BACKEND`, `local`), `dedupe.backend` (`WH_DEDUPE_BACKEND`, `pebble`) and `coord.backend` (`WH_COORD_BACKEND`, `local`). Each layer has only its in-process backend so far, and it is the default, so nothing changes for a config that sets none of them; a value with no backend refuses boot and names the valid ones. A backend's own settings will go in a `.` sub-block; no backend has settings yet, so every such sub-block is an unknown key for now and refuses boot. `internal/app` picks each implementation in one `switch` per layer (`wireMQ`, `wireCache`, `wireDedupe`), `data_dir` is probed only when a selected backend keeps state there (`Config.NeedsDataDir`), and boot logs at `WARN` each line of `Config.Warnings`, the combinations that are correct for one replica only once a shared queue exists. A `config.Config` built without `config.Load` must now name the `mq`, `cache` and `dedupe` backends: the zero value is not the default, and `app.New` refuses it. The defaults live in `defaults()`, like every boot key's since [#631](https://github.com/Wave-RF/WaveHouse/issues/631), so an explicit `backend: ""` in `config.yaml` refuses boot rather than becoming the default. +- **Each layer's implementation is chosen at boot** (`internal/config/{backends,config}.go` (+ tests), `internal/config/defaults_test.go`, `internal/app/{app,wire}.go` (+ tests), `cmd/wavehouse/main.go`, `tests/integration/{setup,tenants}_test.go`, `config.yaml`, `docs/src/content/docs/{configuration.mdx,settings-directory.mdx,architecture.md}`): the first step of running WaveHouse as more than one process ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). The boot config gains `mq.backend` (`WH_MQ_BACKEND`, default `embedded`), `cache.backend` (`WH_CACHE_BACKEND`, `local`), `dedupe.backend` (`WH_DEDUPE_BACKEND`, `pebble`) and `coord.backend` (`WH_COORD_BACKEND`, `local`). Each layer's in-process backend is its default, so nothing changes for a config that sets none of them; a value with no backend refuses boot and names the valid ones. A backend's own settings go in a `.` sub-block (`dedupe.dynamodb` is the first); any other sub-block is an unknown key and refuses boot. `internal/app` picks each implementation in one `switch` per layer (`wireMQ`, `wireCache`, `wireDedupe`), `data_dir` is probed only when a selected backend keeps state there (`Config.NeedsDataDir`), and boot logs at `WARN` each line of `Config.Warnings`, the combinations that are correct for one replica only once a shared queue exists. A `config.Config` built without `config.Load` must now name the `mq`, `cache` and `dedupe` backends: the zero value is not the default, and `app.New` refuses it. The defaults live in `defaults()`, like every boot key's since [#631](https://github.com/Wave-RF/WaveHouse/issues/631), so an explicit `backend: ""` in `config.yaml` refuses boot rather than becoming the default. - **A tenant removed or rejected at runtime has its open streams ended** (`internal/stream/{hub,subscriber,bucket}.go` (+ tests), `internal/api/stream.go` (+ tests), `internal/api/router.go`, `internal/settings/{registry,tree}.go` (+ tests), `internal/ingest/worker.go` (+ tests), `internal/app/wire.go` (+ tests), `clients/ts/src/stream/sse.ts`, `docs/src/content/docs/{api,deployment,architecture,ingest-pipeline}.md`, `docs/src/content/docs/settings-directory.mdx`, `AGENTS.md`): story 3 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). A `GET /v1/stream` used to outlive its tenant: the hub read no policy for it and withheld every row while the keepalive wheel held the connection open, so the client could not tell it from a quiet table. After every reload the hub now evicts the subscribers of each tenant no longer served, removed or rejected alike (`Hub.Prune`, through the close-once `Subscriber.Evict`), and the handler ends the stream: a gap-fill in progress included, and a stream `TenantMW` admitted just before the reload but registered just after it, which the handler checks for as it registers. The client's reconnect gets `404` (removed) or `503` (rejected); the SDK stops on the first and retries the second, resuming from `Last-Event-ID` once the folder is back — except in a browser going cross-origin where tenant `0` is not served or its CORS list does not admit the page, which cannot read either refusal (both are decorated from tenant `0`'s list) and re-dials as after a dropped connection. A flat directory never stops serving tenant `0`, so nothing changes there. Removing a tenant is deleting its folder, then reloading the whole directory, the last folder included: a server started with tenant folders reads the emptied directory as no folder left, where it read as a change of shape and the reload was rejected whole, leaving that tenant served; at boot an empty directory still reads as the four files, missing. Reloading the deleted folder by name leaves its tenant rejected. `GET /v1/health` keeps resolving a tenant, deliberately: its `404` tells a caller with no token no more than every tenant route's does, since they all answer before authenticating, and resolving answers a served tenant's ping from that tenant's own CORS list. A batch queued for a tenant with no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and meets its DLQ switch once, whole, logged once per batch rather than twice per row; a tenant no longer served reads the switch as on, so its queued rows are parked under its own subject rather than dropped, inserted into another tenant's ClickHouse, or left unacked, redelivered for as long as the tenant is away and holding its queue's ack floor, which the purge waits on. Over a nested directory `/livez` no longer keeps naming a tenant that stopped being served before any tenant completed a first discovery: the diagnostic goes back to `no tenant has completed a first discovery yet`. - **One ClickHouse pool per tuple and one schema registry per tenant** (`internal/chconn/chconn.go` (+ tests), `internal/discovery/discovery.go` (+ tests), `internal/app/discoveries.go` (new), `internal/app/{app,wire}.go` (+ tests), `internal/api/{schema,ingest,structured_query,pipes,query,health,errors,clickhouse_exec}.go` (+ tests), `internal/stream/hub.go`, `internal/ingest/worker.go`, `internal/cache/{cache,local,version_manager}.go` (+ tests), `internal/testutil/{testutil,mocks}.go`, `tests/integration/{setup,tenants,boot_resilience,query_limits}_test.go`, `clients/ts/src/{schema,table,sql,client,types}.ts` (+ tests), `tests/e2e/sdk/admin.test.ts`, `docs/src/content/docs/{api,deployment,architecture,ingest-pipeline}.md`, `docs/src/content/docs/{settings-directory,configuration,access-control,reverse-proxy}.mdx`, `docs/src/content/docs/sdk/{admin,reference,queries}.md`, `AGENTS.md`): the second slice of story 6 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files beyond the three noted at the end. The process opens one native pool per distinct `clickhouse.addr` / `database` / `username` / password / `tls` tuple among the served tenants (`chconn.Identity`, `chconn.Pools`), shared by the tenants naming it and sized to their largest `max_open_conns` and `max_idle_conns`; `http_port`, `http_scheme`, `headers` and `query_timeout` stay each tenant's own. Every reload reconciles the pools: a new tuple opens (never dials), a tenant whose tuple changed is repointed, a tuple no tenant names closes after the longest `query_timeout` among the tenants it had, and a pool whose largest ask changed is resized with the same grace. The boot config's `clickhouse.max_total_conns` now bounds the open pools together: boot is refused naming the sum and the ceiling; at a reload a resize above it is refused with the pool kept at its size, and a tuple that cannot be opened — the ceiling, a certificate file that cannot be read, or options the driver refuses — leaves its tenants on the pool they had (the keep-previous-wiring rule of the connection ceiling) or on none when they had none; both are logged and the next reload retries. A tenant on no pool fails closed: `503` with `Retry-After: 30` on `POST /v1/query`, `GET/POST /v1/pipes/{name}`, `POST /v1/ops/query` and `POST /v1/ops/schema/refresh`, before a cached result is served or a query runs. Each served tenant gets a `discovery.SchemaRegistry` of its own over its pool, kept fresh by its own loop — the boot retry until the first success, then `schema.refresh_interval` with the first refresh at a random point within the interval so tenants adopted together do not refresh together — created and stopped from the settings reload and stopped under `App.Close`; `wavehouse_schema_refresh_failures_total{tenant}` counts a loop's failed attempts. `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the strict `?tenant=` the pipe reads take (absent is tenant `0`, `400` malformed, `404` unknown, `503` rejected); the SDK sends it as the `tenant` option of `wh.schema.list()`, `wh.schema.refresh()`, `wh.from(t).schema()` and `wh.sql()`. Over a nested directory `/livez` (and `/v1/health`) is `503` with the latest discovery failure, naming its tenant, while no tenant has completed a first discovery, then `200` for the rest of the process lifetime; `/readyz` pings every open pool at once, is ready at the first answer, and names every pool that did not answer when none does — one tenant's ClickHouse outage is its log line and counter, never a probe failure. The ingest worker inserts each batch into its own tenant's ClickHouse — the tenant the message's topic names, through that tenant's HTTP wiring (`chconn.Pools.Target`); a tenant on no pool takes the failure path an unreachable ClickHouse takes — and its cache invalidation fans out to the tenants on the same address and database as the batch's tenant, whatever their user or `tls` block (they read the same tables), rather than to every known tenant, and a tenant adopted after an absence — rejected or removed, so out of that fan-out — or moved to another address or database has its cached results orphaned in one step (`Cache.InvalidateTenant`, a tenant generation in every version key), so a repaired folder never serves query rows cached before the inserts it missed (since #614 this drops the tenant's cached pipe results too; no insert invalidates a pipe result, which names no table). Three changes reach the single-tenant directory too: a table lookup before the first discovery is a `503` with `Retry-After: 5` (`schema not loaded yet`) rather than a `404`, on `POST /v1/ingest`, `POST /v1/query` and `GET /v1/ops/schema` (the list included, where `[]` would read as no tables); the first periodic refresh fires at a random point within the interval rather than a full interval after boot; and a reload that moves `clickhouse.addr` or `clickhouse.database` now orphans the results cached before it, where they were served until their TTL. Until [#529](https://github.com/Wave-RF/WaveHouse/issues/529) every tenant's user authenticates with `WH_CH_PASSWORD`, so the tuple is in effect the address, database, username and `tls` block. - **One token verifier per tenant, built off the boot and reload paths** (`internal/auth/auth.go` (+ tests), `internal/api/router.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `go.mod`, `docs/src/content/docs/sdk/{reference,streaming}.md`, `docs/src/content/docs/{architecture,deployment,api}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `SECURITY.md`): story 9 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). Over a nested settings directory each tenant's folder now wires that tenant's verifier (`auth.jwks_url`, `auth.role_claim`), so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its identity provider's key set — under any other tenant's header it is refused as invalid — where before one verifier, tenant `0`'s, accepted a token under any header. `auth.Config` is now the boot-config half alone (the HMAC secret and the operator key, shared by every tenant) and the new `auth.Wiring` a tenant's half; `NewAuthenticator` builds no verifier, `Reconfigure(id, wiring)` gives a tenant one — swapped atomically when its wiring changed, kept when it did not — `Prune` drops the verifiers of the tenants a reload stopped serving, rejected or removed alike (no work runs for a tenant that is not served; a folder adopted again gets a fresh verifier), and `Close` stops every JWKS refresh as a component of `App.Close`. This holds on every reload because the registry's `AfterAdopt` hooks run after every reload it applied, empty list included, so a per-tenant reload that rejects a folder drops its verifier then rather than at the next adoption. The middleware reads the request tenant's verifier through an injected `auth.TenantSource` — the store `api.TenantMW` resolved names its tenant with `settings.Store.Tenant()`, one read of the context — a tenant-exempt route (the ops tree) verifies as tenant `0`, and a tenant with no verifier fails closed. The operator key stamps the request tenant's `admin_role`, read from that tenant's policy, rather than tenant `0`'s. A JWKS key set is fetched on its own goroutine: the verifier is in place at once and *pending* until a set has been stored — the first fetch retried with backoff from one second to a minute, a stored set then kept fresh by the library hourly and, rate-limited, on an unknown key id — so neither boot nor a reload (which holds the registry's lock) waits on the endpoint, and **an unreachable JWKS no longer refuses boot** — it logs (`jwks refresh failed; no token validates until it succeeds`) and that tenant alone is affected. A token checked against a pending verifier is a new outcome, `auth.ErrVerifierPending`: every `/v1` route answers it `503 {"error": "token verifier not ready: …"}` with `Retry-After: 30` (`api.refuseUnverifiable`) rather than evaluate the request under the `default_role`, which could accept its data under a lesser role while another pod holding the keys would have served it as its own; a request without a token, and the operator key, are unaffected. Every fetch goes through one client that caps the response at 1 MiB, refusing a larger one as unreachable with `jwks response exceeds 1048576 bytes` as the logged cause. Nothing changes for a flat directory beyond that boot rule: its one tenant gets exactly the verifier it had. The boot warning for the secretless posture (no `auth.jwt_secret` and no `jwks_url`, whose token refusal landed in [#607](https://github.com/Wave-RF/WaveHouse/pull/607)) is now per tenant — each served tenant with no `jwks_url` while the boot secret is unset — where one line that any JWKS tenant silenced used to stand for the whole directory. @@ -31,15 +33,16 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Schema discovery captures each table's DDL, its columns' ordinals and default expressions, and the server version** (`internal/discovery/discovery.go`, `internal/testutil/testutil.go`): `Column` gains `DefaultExpression` and `Position` (both from a widened `system.columns` select), `TableSchema` gains `DDL` from `system.tables.create_table_query`, and `SchemaRegistry` gains `ServerVersion()` from a `SELECT version()` probe next to the existing `SELECT timezone()`. Groundwork for the native type layer, captured on the same refresh as the columns so a stale version cannot outlive the schemas it describes. That is a publication guarantee, not a same-server one: `chconn.Manager` resolves the connection per call, so a reload changing `clickhouse.addr` mid-refresh can still pair a version from one server with schemas from another — narrow, and self-correcting on the next refresh. `DDL` is `json:"-"` and does **not** appear in `/v1/ops/schema`: that endpoint marshals `TableSchema` straight to the client, and an external-engine table (S3, MySQL, PostgreSQL, Kafka) renders its wiring there unconditionally — endpoint, bucket or host, database, username, S3 access key id. ClickHouse masks the password itself as `[HIDDEN]` from ~23.9 (verified on 26.7.3), so the exposure is the topology rather than the secret — except on an older server, or one with `display_secrets_in_show_and_select` enabled. `position` and `default_expression` are additive fields in the response. A table listed in `system.tables` with no `system.columns` rows is skipped rather than published column-less, and both new queries fail the refresh on error exactly as `timezone()` and `system.columns` do — callers keep the prior cache and retry. -- **Settings-directory hot reload — boot loading, three reload triggers, and the config-key migration** (`internal/settings/` (new: `store.go`, `watch.go`, + tests), `internal/api/settings.go` (new, + tests), `internal/api/{router,ingest,structured_query}.go`, `internal/discovery/discovery.go`, `internal/config/config.go`, `cmd/wavehouse/main.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/settings-directory.mdx` (new — the hot-reloadable half of configuration gets its own page; `configuration.mdx` is boot config only); closes the loop [#500](https://github.com/Wave-RF/WaveHouse/pull/500) opened, tracked by [#48](https://github.com/Wave-RF/WaveHouse/issues/48)): the server now *consumes* the settings directory instead of only validating it. `settings.Store` owns the adopted snapshot: `settings.dir` / `WH_SETTINGS_DIR` is now **required**, boot validates and adopts the directory (missing or invalid refuses to start); a running instance then re-validates and re-adopts on any of three triggers — a **directory watch** (fsnotify on the directory, not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost; bursts debounce into one reload), **`SIGHUP`**, and **`POST /v1/ops/settings/reload`** (admin-gated; returns `{"adopted", "findings"}`, `200` adopted / `422` rejected) — all funneling through one serialized reload path. A reload that fails validation keeps the previous good snapshot (an operator mid-edit degrades to a log line, never a broken server); warnings don't block adoption, matching `wavehouse validate`. The tenant tunables **migrate out of boot config** into the directory's `config.json`: `dedupe.id_field` / `dedupe.require_id` (now with the per-table overrides under `dedupe.tables` that [#222](https://github.com/Wave-RF/WaveHouse/issues/222) asked for, resolved per record through the table → global cascade in one atomic snapshot read, so a reload lands at a record boundary and never mixes documents within one record), `query.default_max_rows` and `query.timestamp_bucket_seconds` (read per query), `schema.refresh_interval` (re-read after each tick, so a change applies from the next cycle), `stream.keepalive_interval` / `stream.keepalive_buckets` (a reload calls the new `Heartbeater.Reconfigure`, which rebuilds the keepalive wheel in place with every live subscriber carried over and re-times the running ticker) and `stream.gap_window_minutes` (the sweeper re-reads it every sweep), `mq.max_bytes_gb` (an after-adopt hook updates the tenant's ingest and dead-letter stream limits in place via `mq.Broker.SetMaxBytes` — shrinking below the buffered size backpressures until the sweeper purges it back under the limit, nothing is dropped), `dlq.enabled` with per-table overrides under `dlq.tables` (resolved by the ingest worker at the moment a poison row is isolated: on → park it on the tenant's dead-letter stream and ack; off → leave it unacked for redelivery, never dropped; a served tenant's DLQ stream and `GET /v1/ops/dlq/stats` always exist, so the switch is purely behavioral), the **ClickHouse wiring** (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout`: the new `chconn.Manager` is the one `driver.Conn` every consumer holds and swaps the connection behind it on reload — unconditionally, since the adopted settings are the authority and reachability already surfaces through schema discovery and `/readyz`; the replaced one closes after a `query_timeout` grace; the ingest worker, raw-SQL proxy, and schema registry read the HTTP target, timeout, and database per call), the **auth verifier wiring** (`auth.jwks_url` / `auth.role_claim`: the new `auth.Authenticator` swaps a whole verifier — key source plus its pinned algorithm allowlist — atomically per reload, unconditionally, so an unreachable JWKS fails closed until it can be fetched; `auth.Middleware` is gone — `Authenticator` is the one constructor), and the CORS allowlist (`cors.allowed_origins`, resolved per request). The corresponding YAML/env keys are **removed**: `server.cors_allowed_origins`, `query.default_max_rows`, `schema.refresh_interval`, `dedupe.enabled`, `dedupe.id_field`, `dedupe.require_id`, `stream.keepalive_interval`, `stream.keepalive_buckets`, `mq.gap_window_minutes`, `cache.timestamp_bucket_seconds`, `mq.max_bytes_gb`, `dlq.enabled`, `clickhouse.addr`, `clickhouse.http_port`, `clickhouse.http_scheme`, `clickhouse.database`, `clickhouse.username`, `clickhouse.query_timeout`, `auth.jwks_url`, `auth.role_claim` (and `WH_SERVER_CORS_ALLOWED_ORIGINS`, `WH_QUERY_DEFAULT_MAX_ROWS`, `WH_SCHEMA_REFRESH_INTERVAL`, `WH_DEDUPE_ENABLED`, `WH_DEDUPE_ID_FIELD`, `WH_DEDUPE_REQUIRE_ID`, `WH_STREAM_KEEPALIVE_INTERVAL`, `WH_STREAM_KEEPALIVE_BUCKETS`, `WH_MQ_GAP_WINDOW_MINUTES`, `WH_CACHE_TIMESTAMP_BUCKET_SECONDS`, `WH_MQ_MAX_BYTES_GB`, `WH_DLQ_ENABLED`, `WH_CH_ADDR`, `WH_CH_HTTP_PORT`, `WH_CH_HTTP_SCHEME`, `WH_CH_DATABASE`, `WH_CH_USERNAME`, `WH_CH_QUERY_TIMEOUT`, `WH_AUTH_JWKS_URL`, `WH_AUTH_ROLE_CLAIM`); the secrets — `clickhouse.password`, `auth.jwt_secret`, `auth.operator_key` — stay boot config on purpose (never in a tracked JSON file; combined with the adopted wiring on every reconnect, rotating one is a restart), and boot config is now **strict**: `config.Load` re-reads the YAML against the struct's tags and refuses to start naming every undeclared key, so a `dlq:` or `clickhouse: addr:` left behind can't be read, ignored, and believed; the binary carries **no compiled defaults** — every `config.json` key is required (validation names each missing one), so the adopted snapshot is what the files say, and once adopted it outlives its files (a deleted file or vanished directory is just a rejected reload). Defaults live in one checked-in seed directory (`internal/settings/seed/`, `go:embed`ded): the new **`wavehouse bootstrap [dir]`** writes it (refusing a non-empty directory, the `initdb` contract; the directory resolves exactly as it does for `validate` — the argument, else `WH_SETTINGS_DIR`, usage error with neither — so the two commands are interchangeable on one path and a bare `bootstrap` inside the container images seeds `/app/settings`), the dev `config.yaml` points at a gitignored `./settings` that `make dev` seeds from it, and the e2e fixture ships a copy. The container images ship **no** settings directory: `WH_SETTINGS_DIR` is preset to `/app/settings`, the operator mounts a directory there (`standalone.yaml` bind-mounts the checked-in `deployments/compose/settings/`), and a missing mount refuses to boot rather than running on defaults nobody chose. `dedupe.enabled` moves too: the new `dedupe.Managed` wraps the Pebble store and a `Store.AfterAdopt` hook opens or closes it after every adoption, so flipping the switch is a reload, not a restart (seen ids persist across an off/on cycle; a failed open on reload is logged and ingest fails closed with `500` until the next reload, since the files asked for dedupe — at boot it still refuses to start; a record caught in the instant of the flip is published un-deduped and counted by `wavehouse_ingest_dedupe_disabled_total` rather than failed, and the hook is registered before the boot apply so a reload can never leave the settings and the store out of step). The watcher reloads once as soon as its watch exists, closing the gap between the boot read and the watch — an edit landing in between (a ConfigMap update during a rolling restart) is adopted, not silently missed. `dedupe.enabled` / `WH_DEDUPE_ENABLED` are removed from boot config alongside the other keys. What stays in boot config is only what cannot change under a running process — resource sizing (`data_dir`, `cache.l1_max_cost`), the listeners, the observability exporters — and the secrets. The compose stack now bind-mounts a checked-in `deployments/compose/settings/` (the seed with `clickhouse.addr` pointed at the `clickhouse` service) instead of a volume seeded with `bootstrap`, so the quickstart is `up -d` again; the e2e orchestrator copies the fixture settings per run and patches the testcontainer's ClickHouse ports into `config.json`, since that wiring no longer has an env override. Every after-adopt hook (dedupe, keepalive wheel) is registered before the reload triggers start, so the watcher's first reload can never be missed by a hook. Consumers take functions, not values (`IngestHandler.DedupeSettings`, the structured-query handler's `defaultMaxRows` / `bucketSecs func() int`, the ingest worker's `dlqEnabled func(table) bool`, the sweeper's `gapWindow func() time.Duration`, `corsMiddleware`'s origins getter, `SchemaRegistry`'s database and refresh-interval sources, the query handlers' timeout sources), so `internal/api` stays testable without materializing settings directories. The settings directory is also the **runtime authority for access control and named pipes** (`internal/settings/store.go`, `internal/policy/source.go` (new), `internal/pipes/pipes.go`, `internal/api/{policy,pipes,router}.go`, `internal/stream/hub.go`, `internal/auth/auth.go`, `cmd/wavehouse/main.go`, `Makefile`, `deployments/compose/settings/{policies,roles}.json`, `clients/ts/src/settings.ts` (new); closes [#229](https://github.com/Wave-RF/WaveHouse/issues/229), [#33](https://github.com/Wave-RF/WaveHouse/issues/33), [#461](https://github.com/Wave-RF/WaveHouse/issues/461), [#514](https://github.com/Wave-RF/WaveHouse/issues/514), [#460](https://github.com/Wave-RF/WaveHouse/issues/460), [#363](https://github.com/Wave-RF/WaveHouse/issues/363); advances [#48](https://github.com/Wave-RF/WaveHouse/issues/48) and [#214](https://github.com/Wave-RF/WaveHouse/issues/214)): `roles.json`, `policies.json`, and `pipes.json` are adopted with `config.json` as one snapshot and re-adopted on the same three triggers, and **files are the only write path** — standalone, the operator edits them on the host; on WaveHouse Cloud the control plane writes them — so there is no stored copy that can skip validation: every adoption runs the current rules (strict decode rejecting unknown and duplicate keys, the full policy validation including the claim-template grammar, pipe name/SQL/parameter-type rules, and the cross-file check that every role a grant or `allowed_roles` names is declared in `roles.json`), and a rejected edit keeps the previous good policy and pipes in effect. `policies.json` is one policy document (`{}` = no policy, adopted fail-closed with a warning); `pipes.json` carries full definitions (`allowed_roles`, `parameters`, `description`), so a file-defined pipe is no longer admin-only by construction. Consumers read the adopted snapshot per request through `policy.Source` (a `func() *policy.Policy`; `settings.Store.Policy` in production, `policy.Static(p)` in tests) and `pipes.Source` (`settings.Store`; `pipes.Static(q...)` in tests), so a reload applies to the very next request, including the SSE hub's per-event policy read. `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, and pipe execution are unchanged; the operator key still passes the `/v1/ops/*` gate under no policy, now as the break-glass that inspects the policy and triggers `POST /v1/ops/settings/reload` after `policies.json` is fixed. The SDK gains `wh.settings.reload()` (`POST /v1/ops/settings/reload`, returning `{ adopted, findings }`). The compose stack's trial `public` policy moves into the bind-mounted `deployments/compose/settings/policies.json` + `roles.json`, and `make dev` copies the same two files into its seeded `./settings` so a fresh dev server works tokenless. **Removed** — the write endpoints `PUT /v1/ops/policy`, `PUT /v1/ops/pipes/{name}`, and `DELETE /v1/ops/pipes/{name}`; the NATS KV buckets `WAVEHOUSE_POLICY` and `WAVEHOUSE_PIPES` and their KV Watch sync (`internal/policy/store.go`, the pipes KV store); the boot-config keys `policy.file_path` / `WH_POLICY_FILE_PATH` and `pipes.dir` / `WH_PIPES_DIR` (a leftover `policy:` or `pipes:` YAML block now refuses boot by name, like the other moved keys) and the `.sql`-directory pipes bootstrap; `deployments/compose/dev-policy.yaml`; the SDK methods `wh.policy.set`, `wh.pipes.set`, and `wh.pipes.delete`; and the test helpers `policy.NewMemoryStore`, `pipes.NewMemoryStore`, and `testutil/natsjs.go`. +- **Settings-directory hot reload — boot loading, three reload triggers, and the config-key migration** (`internal/settings/` (new: `store.go`, `watch.go`, + tests), `internal/api/settings.go` (new, + tests), `internal/api/{router,ingest,structured_query}.go`, `internal/discovery/discovery.go`, `internal/config/config.go`, `cmd/wavehouse/main.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/settings-directory.mdx` (new — the hot-reloadable half of configuration gets its own page; `configuration.mdx` is boot config only); closes the loop [#500](https://github.com/Wave-RF/WaveHouse/pull/500) opened, tracked by [#48](https://github.com/Wave-RF/WaveHouse/issues/48)): the server now *consumes* the settings directory instead of only validating it. `settings.Store` owns the adopted snapshot: `settings.dir` / `WH_SETTINGS_DIR` is now **required**, boot validates and adopts the directory (missing or invalid refuses to start); a running instance then re-validates and re-adopts on any of three triggers — a **directory watch** (fsnotify on the directory, not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost; bursts debounce into one reload), **`SIGHUP`**, and **`POST /v1/ops/settings/reload`** (admin-gated; returns `{"adopted", "findings"}`, `200` adopted / `422` rejected) — all funneling through one serialized reload path. A reload that fails validation keeps the previous good snapshot (an operator mid-edit degrades to a log line, never a broken server); warnings don't block adoption, matching `wavehouse validate`. The tenant tunables **migrate out of boot config** into the directory's `config.json`: `dedupe.id_field` / `dedupe.require_id` (now with the per-table overrides under `dedupe.tables` that [#222](https://github.com/Wave-RF/WaveHouse/issues/222) asked for, resolved per record through the table → global cascade in one atomic snapshot read, so a reload lands at a record boundary and never mixes documents within one record), `query.default_max_rows` and `query.timestamp_bucket_seconds` (read per query), `schema.refresh_interval` (re-read after each tick, so a change applies from the next cycle), `stream.keepalive_interval` / `stream.keepalive_buckets` (a reload calls the new `Heartbeater.Reconfigure`, which rebuilds the keepalive wheel in place with every live subscriber carried over and re-times the running ticker) and `stream.gap_window_minutes` (the sweeper re-reads it every sweep), `mq.max_bytes_gb` (an after-adopt hook updates the tenant's ingest and dead-letter stream limits in place via `mq.Broker.SetMaxBytes` — shrinking below the buffered size backpressures until the sweeper purges it back under the limit, nothing is dropped), `dlq.enabled` with per-table overrides under `dlq.tables` (resolved by the ingest worker at the moment a poison row is isolated: on → park it on the tenant's dead-letter stream and ack; off → leave it unacked for redelivery, never dropped; a served tenant's DLQ stream and `GET /v1/ops/dlq/stats` always exist, so the switch is purely behavioral), the **ClickHouse wiring** (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout`: the new `chconn.Manager` is the one `driver.Conn` every consumer holds and swaps the connection behind it on reload — unconditionally, since the adopted settings are the authority and reachability already surfaces through schema discovery and `/readyz`; the replaced one closes after a `query_timeout` grace; the ingest worker, raw-SQL proxy, and schema registry read the HTTP target, timeout, and database per call), the **auth verifier wiring** (`auth.jwks_url` / `auth.role_claim`: the new `auth.Authenticator` swaps a whole verifier — key source plus its pinned algorithm allowlist — atomically per reload, unconditionally, so an unreachable JWKS fails closed until it can be fetched; `auth.Middleware` is gone — `Authenticator` is the one constructor), and the CORS allowlist (`cors.allowed_origins`, resolved per request). The corresponding YAML/env keys are **removed**: `server.cors_allowed_origins`, `query.default_max_rows`, `schema.refresh_interval`, `dedupe.enabled`, `dedupe.id_field`, `dedupe.require_id`, `stream.keepalive_interval`, `stream.keepalive_buckets`, `mq.gap_window_minutes`, `cache.timestamp_bucket_seconds`, `mq.max_bytes_gb`, `dlq.enabled`, `clickhouse.addr`, `clickhouse.http_port`, `clickhouse.http_scheme`, `clickhouse.database`, `clickhouse.username`, `clickhouse.query_timeout`, `auth.jwks_url`, `auth.role_claim` (and `WH_SERVER_CORS_ALLOWED_ORIGINS`, `WH_QUERY_DEFAULT_MAX_ROWS`, `WH_SCHEMA_REFRESH_INTERVAL`, `WH_DEDUPE_ENABLED`, `WH_DEDUPE_ID_FIELD`, `WH_DEDUPE_REQUIRE_ID`, `WH_STREAM_KEEPALIVE_INTERVAL`, `WH_STREAM_KEEPALIVE_BUCKETS`, `WH_MQ_GAP_WINDOW_MINUTES`, `WH_CACHE_TIMESTAMP_BUCKET_SECONDS`, `WH_MQ_MAX_BYTES_GB`, `WH_DLQ_ENABLED`, `WH_CH_ADDR`, `WH_CH_HTTP_PORT`, `WH_CH_HTTP_SCHEME`, `WH_CH_DATABASE`, `WH_CH_USERNAME`, `WH_CH_QUERY_TIMEOUT`, `WH_AUTH_JWKS_URL`, `WH_AUTH_ROLE_CLAIM`); the secrets — `clickhouse.password`, `auth.jwt_secret`, `auth.operator_key` — stay boot config on purpose (never in a tracked JSON file; combined with the adopted wiring on every reconnect, rotating one is a restart), and boot config is now **strict**: `config.Load` re-reads the YAML against the struct's tags and refuses to start naming every undeclared key, so a `dlq:` or `clickhouse: addr:` left behind can't be read, ignored, and believed; the binary carries **no compiled defaults** but one — every `config.json` key except `dedupe.retention` (missing means `"0"`, forever) is required (validation names each missing one), so the adopted snapshot is what the files say, and once adopted it outlives its files (a deleted file or vanished directory is just a rejected reload). Defaults live in one checked-in seed directory (`internal/settings/seed/`, `go:embed`ded): the new **`wavehouse bootstrap [dir]`** writes it (refusing a non-empty directory, the `initdb` contract; the directory resolves exactly as it does for `validate` — the argument, else `WH_SETTINGS_DIR`, usage error with neither — so the two commands are interchangeable on one path and a bare `bootstrap` inside the container images seeds `/app/settings`), the dev `config.yaml` points at a gitignored `./settings` that `make dev` seeds from it, and the e2e fixture ships a copy. The container images ship **no** settings directory: `WH_SETTINGS_DIR` is preset to `/app/settings`, the operator mounts a directory there (`standalone.yaml` bind-mounts the checked-in `deployments/compose/settings/`), and a missing mount refuses to boot rather than running on defaults nobody chose. `dedupe.enabled` moves too: the new `dedupe.Managed` wraps the Pebble store and a `Store.AfterAdopt` hook opens or closes it after every adoption, so flipping the switch is a reload, not a restart (seen ids persist across an off/on cycle; a failed open on reload is logged and ingest fails closed with `500` until the next reload, since the files asked for dedupe — at boot it still refuses to start; a record caught in the instant of the flip is published un-deduped and counted by `wavehouse_ingest_dedupe_disabled_total` rather than failed, and the hook is registered before the boot apply so a reload can never leave the settings and the store out of step). The watcher reloads once as soon as its watch exists, closing the gap between the boot read and the watch — an edit landing in between (a ConfigMap update during a rolling restart) is adopted, not silently missed. `dedupe.enabled` / `WH_DEDUPE_ENABLED` are removed from boot config alongside the other keys. What stays in boot config is only what cannot change under a running process — resource sizing (`data_dir`, `cache.l1_max_cost`), the listeners, the observability exporters — and the secrets. The compose stack now bind-mounts a checked-in `deployments/compose/settings/` (the seed with `clickhouse.addr` pointed at the `clickhouse` service) instead of a volume seeded with `bootstrap`, so the quickstart is `up -d` again; the e2e orchestrator copies the fixture settings per run and patches the testcontainer's ClickHouse ports into `config.json`, since that wiring no longer has an env override. Every after-adopt hook (dedupe, keepalive wheel) is registered before the reload triggers start, so the watcher's first reload can never be missed by a hook. Consumers take functions, not values (`IngestHandler.DedupeSettings`, the structured-query handler's `defaultMaxRows` / `bucketSecs func() int`, the ingest worker's `dlqEnabled func(table) bool`, the sweeper's `gapWindow func() time.Duration`, `corsMiddleware`'s origins getter, `SchemaRegistry`'s database and refresh-interval sources, the query handlers' timeout sources), so `internal/api` stays testable without materializing settings directories. The settings directory is also the **runtime authority for access control and named pipes** (`internal/settings/store.go`, `internal/policy/source.go` (new), `internal/pipes/pipes.go`, `internal/api/{policy,pipes,router}.go`, `internal/stream/hub.go`, `internal/auth/auth.go`, `cmd/wavehouse/main.go`, `Makefile`, `deployments/compose/settings/{policies,roles}.json`, `clients/ts/src/settings.ts` (new); closes [#229](https://github.com/Wave-RF/WaveHouse/issues/229), [#33](https://github.com/Wave-RF/WaveHouse/issues/33), [#461](https://github.com/Wave-RF/WaveHouse/issues/461), [#514](https://github.com/Wave-RF/WaveHouse/issues/514), [#460](https://github.com/Wave-RF/WaveHouse/issues/460), [#363](https://github.com/Wave-RF/WaveHouse/issues/363); advances [#48](https://github.com/Wave-RF/WaveHouse/issues/48) and [#214](https://github.com/Wave-RF/WaveHouse/issues/214)): `roles.json`, `policies.json`, and `pipes.json` are adopted with `config.json` as one snapshot and re-adopted on the same three triggers, and **files are the only write path** — standalone, the operator edits them on the host; on WaveHouse Cloud the control plane writes them — so there is no stored copy that can skip validation: every adoption runs the current rules (strict decode rejecting unknown and duplicate keys, the full policy validation including the claim-template grammar, pipe name/SQL/parameter-type rules, and the cross-file check that every role a grant or `allowed_roles` names is declared in `roles.json`), and a rejected edit keeps the previous good policy and pipes in effect. `policies.json` is one policy document (`{}` = no policy, adopted fail-closed with a warning); `pipes.json` carries full definitions (`allowed_roles`, `parameters`, `description`), so a file-defined pipe is no longer admin-only by construction. Consumers read the adopted snapshot per request through `policy.Source` (a `func() *policy.Policy`; `settings.Store.Policy` in production, `policy.Static(p)` in tests) and `pipes.Source` (`settings.Store`; `pipes.Static(q...)` in tests), so a reload applies to the very next request, including the SSE hub's per-event policy read. `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, and pipe execution are unchanged; the operator key still passes the `/v1/ops/*` gate under no policy, now as the break-glass that inspects the policy and triggers `POST /v1/ops/settings/reload` after `policies.json` is fixed. The SDK gains `wh.settings.reload()` (`POST /v1/ops/settings/reload`, returning `{ adopted, findings }`). The compose stack's trial `public` policy moves into the bind-mounted `deployments/compose/settings/policies.json` + `roles.json`, and `make dev` copies the same two files into its seeded `./settings` so a fresh dev server works tokenless. **Removed** — the write endpoints `PUT /v1/ops/policy`, `PUT /v1/ops/pipes/{name}`, and `DELETE /v1/ops/pipes/{name}`; the NATS KV buckets `WAVEHOUSE_POLICY` and `WAVEHOUSE_PIPES` and their KV Watch sync (`internal/policy/store.go`, the pipes KV store); the boot-config keys `policy.file_path` / `WH_POLICY_FILE_PATH` and `pipes.dir` / `WH_PIPES_DIR` (a leftover `policy:` or `pipes:` YAML block now refuses boot by name, like the other moved keys) and the `.sql`-directory pipes bootstrap; `deployments/compose/dev-policy.yaml`; the SDK methods `wh.policy.set`, `wh.pipes.set`, and `wh.pipes.delete`; and the test helpers `policy.NewMemoryStore`, `pipes.NewMemoryStore`, and `testutil/natsjs.go`. - **"Was this page helpful?" feedback widget on every docs page** (`docs/src/components/PageFeedback.astro` (new), `docs/src/components/Footer.astro`): a thumbs-up / thumbs-down vote below the page content, captured to PostHog as `docs_feedback` with `{ helpful, page }`. It renders from `Footer.astro`'s sidebar branch — the same indirection the Cloud CTA uses — rather than a per-page import or frontmatter flag, so every content page gets it automatically, including ones not written yet; it sits *below* the Cloud CTA on the pages that carry one, and splash pages (the homepage and 404) take the other footer branch and never render it. One vote per page per visitor: the choice is remembered in `localStorage` keyed by pathname, and a revisit renders the thanks message instead of re-prompting (storage is a nicety, not the record — a browser with storage disabled still votes). - **Settings-directory validation — `wavehouse validate [dir]`** (`internal/settings/` (new: `settings.go`, `validate.go`, `decode.go`, `finding.go`, + tests), `cmd/wavehouse/validate.go` (new, + tests), `cmd/wavehouse/main.go`): first piece of the file-based control plane (settings live in a directory of JSON documents — `roles.json`, `policies.json`, `pipes.json`, `config.json` — that a running instance will hot-reload; this change is validation-only — boot loading and reload wiring land separately). `settings.Validate(dir)` is the single gate every consumer of the directory runs: deliberately pure (no network, no ClickHouse — table/column existence stays with schema discovery, per Bring-Your-Own-Schema), and it collects **all** findings in one pass instead of failing on the first. Checks, layered: the directory holds exactly the four files (a missing file is an error — an empty document is `{}`, so absence always means deletion or a wrong path; any unexpected entry — file or directory — is an error so a typoed `polices.json` or a stray backup can't be silently ignored; dot-prefixed entries are the one carve-out, since erroring on vim swap files or the `..data` machinery Kubernetes ConfigMap mounts publish through would break hand editing and the cloud fan-out's mount pattern alike); strict JSON syntax (unknown fields rejected — the JSON form of the retired-config-key trap; empty/truncated files rejected, never read as an empty document; a leading UTF-8 byte order mark named as such instead of surfacing as a cryptic invalid-character error; a directory, unreadable file, or non-regular file (a FIFO would hang the read forever waiting for a writer; a stat gate rejects it — following symlinks, so Kubernetes ConfigMap mounts' symlink layout still passes) squatting on a settings filename named as the one real problem, not double-reported as "missing"; a top-level `null` rejected — the one well-formed document that decodes into a zero value without error, so it would silently read as "no settings"; trailing content rejected; duplicated object keys detected by a token-level pass, since `encoding/json` silently keeps the last copy); per-file shape rules (role names non-empty/unique, pipe names/SQL/param types, `config.json` bounds mirroring boot-config validation — its sections are the *tenant-owned* behavioral tunables (dedupe id_field/require_id plus per-table overrides under `dedupe.tables` — each entry overrides only the fields it names, resolving table → global → compiled default per field, so the effective id_field can never be empty — an explicit empty, whitespace-only, or whitespace-padded id_field is rejected at both levels, since an exact-match JSON key lookup would silently miss every row ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s shape, unblocked by the file design since table names are runtime-resolved like policy grants); query default_max_rows, schema refresh_interval, CORS origins); platform-owned knobs like the SSE keepalives deliberately stay boot config); and cross-file referential integrity (every role a policy grant, `default_role`/`admin_role`, or pipe allowlist references must be declared in `roles.json`; an empty role string in a grant or allowlist is named as such — it matches no request and authorizes nobody). Warnings don't invalidate: a grant scoping the admin role (an unconditional bypass — dead config), `default_role` = admin, and a `default` on a required pipe parameter are flagged but legal. An empty `policies.json` means no policy — fail closed, matching deleted-policy semantics — and draws a warning naming the total lockout, so it announces itself at validation time instead of one 403 at a time. The CLI (`cmd/wavehouse/validate.go`, following the `health` subcommand pattern) takes the directory as an argument or from `WH_SETTINGS_DIR`, prints findings, and exits 0/1/2 (valid/invalid/usage) so CI and operators can gate config changes before they reach a running instance. The dispatch in `main.go` also grows `help` and `version` subcommands, and an unknown command is now a usage error instead of silently falling through and starting the server (`wavehouse validat` booting a listener is not a typo anyone wants); each subcommand parses its arguments with a stdlib `flag.FlagSet`, so `wavehouse -h` prints command-specific help and a stray flag or argument is a usage error rather than being silently swallowed. `WH_SETTINGS_DIR` has a single authority: `config.EnvSettingsDir`, with a reflection test pinning the `settings.dir` struct tag to it. The directory's location joins boot config as `settings.dir` (`WH_SETTINGS_DIR`; `internal/config/config.go`, `config.yaml`, `docs/src/content/docs/configuration.mdx`) — boot-tier by necessity, since it's the pointer the reload machinery follows; no default, same silent-misconfiguration reasoning as `policy.file_path`. - **Docs-site analytics for search, code copies, 404s, docs section, and live-demo connectivity** (`docs/src/components/DocsTracking.astro` (new), `docs/src/components/{PostHog,Footer,LiveDemo}.astro`): the site tracked its own CTAs but nothing a reader did on the way to one, so the questions that decide what to write next — what people search for and *don't* find, which snippets get copied, which dead links keep getting followed — had no data behind them. `docs_search` fires a second after the query settles rather than once per keystroke, carrying `query` and `result_count` read off Pagefind's own results message (the rendered list is capped at its page size, so counting the DOM would under-report); `result_count: 0` is the event worth having. `code_copied` (`page`, `language`) watches Expressive Code's copy buttons from the document rather than re-binding every code block on every navigation — the hero's install chip is not an EC block and keeps its own `hero_install_copied`. `docs_404` (`path`, `referrer`) turns broken inbound links into a list instead of a hunch. A `doc_section` property (the first path segment, `home` for `/`) puts every event in a docs area without each tracker carrying its own copy; it's stamped at capture time by a `before_send` hook in `posthog.init()` rather than `register()`, because a queued `register()` replays only after init has already captured the first hard-load `$pageview` — which would then carry the previous visit's persisted value — and `history_change` navigations update the URL before capture fires, so reading `location` in the hook is always current. `live_demo_connected` fires once per mount when the hero's SSE feed comes up rather than on its first row — named for what it measures (the demo backend answered), since a quiet minute on the repo is not a disengaged reader. The three site-wide trackers share one new `DocsTracking.astro` rendered from the footer (like `MermaidZoom` / `ScrollHints`) and delegate from `document`, since Pagefind, Expressive Code, and the 404 route all own their own markup — some of it created after page load. +- **Dedupe retention per tenant and table, and a sweep that deletes expired ids** (`internal/settings/{settings,validate,store}.go` (+ tests), `internal/settings/seed/config.json`, `internal/dedupe/{embedded,sweep}.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/testutil/mocks.go`, `deployments/compose/settings/config.json`, `tests/e2e/fixtures/settings/config.json`, `config.yaml`, `docs/src/content/docs/{settings-directory.mdx,deployment,durability,architecture}.md`, `AGENTS.md`): [#220](https://github.com/Wave-RF/WaveHouse/issues/220). `config.json` gains an optional **`dedupe.retention`** key, overridable per table in `dedupe.tables.
.retention`: how long a committed id stays a duplicate, as a duration string (`"720h"`), or `"0"` to keep it forever, which is the seed value and the behaviour before this release. A `config.json` without the key keeps ids forever, and a table override without one inherits the tenant's, so an existing directory needs no change. A finite retention below `"2m"`, the ingest queue's duplicate window, is refused rather than raised to the minimum: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and dropped by the queue as a copy while the client was told it was accepted. It is hot-reloadable and read per record like `id_field`; a change applies to ids committed after it, and a reload landing mid-window commits each record with the retention it was prepared under. The embedded Pebble store already treated an expired id as new; it now also deletes expired keys, and the version-0 keys the key-layout change left behind, in a background sweep that starts a minute after the instance opens and repeats hourly. It reads 1,024 keys per chunk without a lock, then re-reads the expired and version-0 ones under a lock `Commit` also takes and deletes, without fsync, those that still are, so an id committed again after the sweep read it is never deleted, and a `Commit` waits for at most one chunk's re-reads, never for the deleted keys a chunk steps over. New metric `wavehouse_dedupe_swept_keys_total{reason="expired"|"version_0"}`. `settings.Store.DedupeFor` now returns a `settings.Dedupe` struct rather than three values. ### Changed -- **One escaping for composite keys, `-` kept; dead-letter counts per table** (`internal/keyenc` (new, + tests), `internal/mq/{mq,subject,embedded,deadletter}.go` (+ tests), `internal/query/ident.go` (+ tests), `AGENTS.md`, `docs/src/content/docs/{architecture,development}.md`): part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). NATS subject tokens and the cache's namespace tokens each carried a copy of the same encoder; both now call `internal/keyenc`, which the dedupe keys will use too, and subjects are built with its `Join` (each field escaped, with a separator the escaping never writes between them). The escaping now keeps `-` as well as ASCII letters, digits and `_` — exactly the tenant-id grammar — so a table or scope such as `my-table` is `my-table` in a subject rather than `my%2Dtable`; every other byte is escaped as before (pinned by golden tests, and against v0.1.0's encoder for every other byte value). Upgrading from v0.1.0 notices nothing further, since its queue is deleted at boot (below). A queue an unreleased build since [#612](https://github.com/Wave-RF/WaveHouse/pull/612) wrote still reads, because decoding is `url.PathUnescape` as it was: `%2D` decodes to `-`, and a dead-letter count merges both forms. Only a `/v1/stream` client resuming across such an upgrade (`Last-Event-ID` or `since`) on a table whose name holds `-` misses that table's events queued before it, since the replay filters on the table's exact subject. `GET /v1/ops/dlq/stats` now counts every scope of a table under the table itself, and `?table=` keeps all of its scopes; a scoped message used to count under `table.scope`, a name a dotted table could share. Scope is always empty today, so the response is unchanged. +- **One escaping for composite keys, `-` kept; dead-letter counts per table** (`internal/keyenc` (new, + tests), `internal/mq/{mq,subject,embedded,deadletter}.go` (+ tests), `internal/query/ident.go` (+ tests), `AGENTS.md`, `docs/src/content/docs/{architecture,development}.md`): part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). NATS subject tokens and the cache's namespace tokens each carried a copy of the same encoder; both now call `internal/keyenc`, which the dedupe keys use too, and subjects are built with its `Join` (each field escaped, with a separator the escaping never writes between them). The escaping now keeps `-` as well as ASCII letters, digits and `_` — exactly the tenant-id grammar — so a table or scope such as `my-table` is `my-table` in a subject rather than `my%2Dtable`; every other byte is escaped as before (pinned by golden tests, and against v0.1.0's encoder for every other byte value). Upgrading from v0.1.0 notices nothing further, since its queue is deleted at boot (below). A queue an unreleased build since [#612](https://github.com/Wave-RF/WaveHouse/pull/612) wrote still reads, because decoding is `url.PathUnescape` as it was: `%2D` decodes to `-`, and a dead-letter count merges both forms. Only a `/v1/stream` client resuming across such an upgrade (`Last-Event-ID` or `since`) on a table whose name holds `-` misses that table's events queued before it, since the replay filters on the table's exact subject. `GET /v1/ops/dlq/stats` now counts every scope of a table under the table itself, and `?table=` keeps all of its scopes; a scoped message used to count under `table.scope`, a name a dotted table could share. Scope is always empty today, so the response is unchanged. - **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch and the hub bridge's fetch-ahead are each shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it, and so is the tracking of tenant `0`'s last adopted store: a flat directory's ops gate, its one reader left, reads tenant `0` through the registry). One tenant's failed purge holds up no other tenant's, and the sweep logs it at `ERROR` unless every failure in it is a buffer consumer not created yet. A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each reload trying again, and a publish at most once every five seconds, so its clients retrying never hold up another tenant's queue. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. @@ -87,6 +90,9 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Fixed +- **Dedupe claims an id, publishes, then commits it — and keys it by tenant, table and id** (`internal/dedupe/{dedupe,key,embedded,managed}.go` (+ tests), `internal/dedupe/dedupetest/` (new), `internal/api/ingest.go` (+ tests), `internal/settings/validate_test.go`, `internal/keyenc/keyenc.go`, `internal/testutil/mocks.go`, `internal/app/app_test.go`, `AGENTS.md`, `docs/src/content/docs/{architecture,api,deployment,development}.md`, `settings-directory.mdx`, `sdk/reference.md`): `CheckAndMark` is replaced by a two-phase `Reserve` → `Commit` / `Release` contract with a lease on the pending claim, and every backend now runs one conformance suite. Four bugs go with it. Two concurrent requests carrying one id no longer both publish it: Pebble's check and claim happen under one lock, and the loser answers `503` with `Retry-After` while the winner is still publishing ([#390](https://github.com/Wave-RF/WaveHouse/issues/390)). A publish that fails gives its id back, so the retry a `503` asks for is published instead of skipped as a duplicate of a record that never reached the queue ([#384](https://github.com/Wave-RF/WaveHouse/issues/384)) — the residual case is a publish that fails after it already reached the broker (a timeout, a disconnect), where the released id lets the retry through but that retry publishes a genuine second copy; [#629](https://github.com/Wave-RF/WaveHouse/pull/629) closes that with an idempotency key. The same id in two tables is two ids ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s keyspace half). An explicit `null` id is a missing id — rejected under `require_id`, published un-deduped otherwise — instead of the one id `""` that made every null record after the first a duplicate ([#370](https://github.com/Wave-RF/WaveHouse/issues/370)). **Upgrade:** the key layout changes, so an id seen before the upgrade is accepted once more after it; nothing is migrated, and the old keys are left in `/pebble`, unread, deleted by the retention sweep (see Added) ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)) ([Deployment → Upgrading across the dedupe key change](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md#upgrading-across-the-dedupe-key-change)). The key is readable text, `/
/` (for example `acme/clicks/evt-123`), with the table and id escaped and joined by `internal/keyenc`, the escaping NATS subject tokens already use, so any table name gets a keyspace of its own, including one holding a NUL byte or a `/`. New metrics: `wavehouse_ingest_dedupe_commit_failed_total` (a published record whose id failed to commit; the claim lapses with its lease) and `wavehouse_dedupe_hashed_id_total` (an id over 1,024 bytes once escaped, stored as its SHA-256). +- **Ingest runs in windows of 256 records over the dedupe contract, and a dedupe store that cannot answer is a `503`** (`internal/api/ingest.go` (+ tests), `internal/mq/{mq,embedded}.go` (+ tests), `internal/dedupe/key.go` (+ tests), `internal/testutil/mocks.go`, `AGENTS.md`, `docs/src/content/docs/{api,architecture,durability}.md`, `settings-directory.mdx`, `sdk/reference.md`): each window of a request is prepared, then reserved in one dedupe call, published in order, and committed in one call, so a batch costs one dedupe round trip per phase per window rather than per record — on Pebble, one commit `fsync` per window (a 1,000-record batch: four syncs instead of a thousand, 24 ms against 5.7 s of dedupe time measured with the queue stubbed). Every deduped record is published under an idempotency key (`mq.WithIdempotencyKey`, JetStream's message id, derived by `dedupe.IdempotencyKey`), and each tenant's ingest stream now keeps an explicit two-minute duplicate window, sized to `2 × the 30-second lease + 1s`: an uncertain publish's `503` sends the *full* lease as `Retry-After`, so an obedient client's retry can land up to ~2×lease after the original request, and the `+1s` covers a backend whose claim expiry itself rounds up by that much. That closes the last path of [#384](https://github.com/Wave-RF/WaveHouse/issues/384): a publish that fails with an unknown outcome (anything but a full queue) keeps its record's claim until the lease lapses instead of releasing it, and a retry after the lease but within two minutes of the first publish is dropped by the queue if the first copy was stored (a later one is stored again). A dedupe store that is not open or that reports `dedupe.ErrUnavailable` now answers `503 {"error":"dedupe store unavailable"}` with `Retry-After: 5`, which the SDK retries, rather than `500 dedupe failed`. A mid-body read error or dedupe failure now drops the open window unpublished, where records before it used to be published; `wavehouse_ingest_dedupe_commit_failed_total` and `wavehouse_ingest_dedupe_disabled_total` count records, as before, now added a window at a time. +- **Tests that start the embedded broker no longer fail removing its store after passing** (`internal/testutil/storedir` (new, + tests), `internal/testutil/testutil.go`, `internal/mq/embedded.go` (comment), `internal/mq/{embedded_test,mqtest/embedded_test}.go`, `internal/ingest/worker_test.go`, `internal/app/{app,roles}_test.go`, `cmd/wavehouse/main_test.go`, `tests/integration/{ingest_outage,query_errors,tenants}_test.go`, `AGENTS.md`, `docs/src/content/docs/development.md`): [#442](https://github.com/Wave-RF/WaveHouse/issues/442). The NATS server writes each durable consumer's state (`obs//o.dat`, through a temporary file renamed into place) from a goroutine that neither `Shutdown` nor `WaitForShutdown` joins, and its consumer store waits for that goroutine at close only when state is still unwritten, for at most 100ms — so a write already under way lands after `EmbeddedNATS.Close` returns, and `t.TempDir`'s one-shot `RemoveAll` met the late entry as `directory not empty`. Under parallel test processes it failed about 4% of the ingest worker tests (78 of 1,800 runs). Every store a test puts on disk now comes from `storedir.New(t)`, whose cleanup — after the broker's `Close` — removes it again whenever a directory was refilled between being read and being removed: each late write adds at most two entries and none once its directory is gone, so the removal ends without a timer (0 of 1,800 under the same load). It replaces two sleep-and-retry copies in the `internal/mq` tests. `TestStartIngestWorker_StopFunc_RespectsShutdownDeadline` also joins the worker its deadline abandons before the broker closes, rather than leaving it to ack on a closed connection. - **A pipe that writes runs on every call instead of being answered from the cache** (`internal/api/{pipes,ch_errors}.go` (+ tests), `docs/src/content/docs/{pipes.mdx,api.md,architecture.md,configuration.mdx,settings-directory.mdx,ingest-pipeline.md,sdk/pipes.md,sdk/reference.md}`, `clients/ts/src/pipes.ts` (doc comment), `internal/{settings/settings,app/wire}.go` (comments), `AGENTS.md`): fixes [#386](https://github.com/Wave-RF/WaveHouse/issues/386). `/v1/pipes/{name}` sent a write's SQL to ClickHouse through `Exec`, but still cached the `[]` it returned and coalesced identical calls in flight, so a repeat within the TTL answered `200` without executing and concurrent identical calls became one write — silently dropped writes, and with a shared cache ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)) on every instance. A pipe whose bound SQL `IsMutation` classifies as a write — the same classifier that picks `Exec` — now skips the cache lookup, the fill and singleflight, and answers `X-Cache: BYPASS` with `Cache-Control: no-store`, so an HTTP cache in front of a `GET` cannot drop the write either. Classification stays automatic rather than a declared pipe property, so an operator cannot forget to mark one, and costs no ClickHouse round trip. A failed write answers with the status and `code` a failed read gets (see the ClickHouse-errors entry below), but always `retryable: false` and with no `Retry-After`, `503 clickhouse.unavailable` included: the statement may have run, so the SDK does not retry it. A write refused before it is sent, the tenant on no pool, keeps its `503` with `Retry-After: 30`. Read pipes are unchanged. Not in this fix: a write pipe still does not invalidate cached reads of the table it writes ([#394](https://github.com/Wave-RF/WaveHouse/issues/394)). - **The write classifier skips whitespace, comments and quoted text the way ClickHouse's lexer does, classifies a `WITH`-led statement by `INSERT INTO` alone, and looks through `EXECUTE AS`** (`internal/api/clickhouse_exec.go` (+ tests), `internal/testutil/mutationtest` (new), `tests/integration/ismutation_test.go` (new), `docs/src/content/docs/pipes.mdx`, `AGENTS.md`): `IsMutation` picks `Exec` for a write, and since [#386](https://github.com/Wave-RF/WaveHouse/issues/386) keeps a write pipe out of the cache. It missed a write behind a backslash-escaped quote (`'it\'s'`, and the same inside `"…"` and `` `…` ``), a heredoc (`$$ ( $$`, `$tag$ … $tag$`), a curly-quoted literal or identifier (`‘(’`, `“c(d”`), a `//` line comment, a nested block comment (`/* a /* b */ SELECT */ INSERT …`), a number led by `.` with the verb glued to it (`WITH 1 AS a, .5INSERT INTO t …`, which ClickHouse reads as `.5` then `INSERT`), an `EXECUTE AS ` prefix (`EXECUTE AS u INSERT …`), or leading whitespace other than space, tab, CR and LF: `\v`, `\f`, a no-break space, a byte-order mark, and the other Unicode spaces ClickHouse skips. A missed write went through `Query`, which ran it and then failed the call with a `5xx` the TypeScript SDK retries, so one call could write three times. The same gaps, and a word led by `_` (`_delete`) whose tail was read as a verb, could make a read look like a write, which runs through `Exec` and answers `[]`. After a `WITH` list, which ClickHouse follows only with `SELECT`, a FROM-first `SELECT` or `INSERT INTO`, a name spelled like a keyword was taken for the statement: `WITH 'd' AS desc INSERT …` and `WITH 1 AS select INSERT …` ran as reads, and `WITH 1 AS set SELECT set` and `WITH 1 AS x FROM system.one SELECT x` as writes. A `WITH`-led statement is now a write exactly when it holds `INSERT INTO` outside parentheses. The classifier, exported as `IsMutation` for it, is now checked against the pinned ClickHouse's own parser (`EXPLAIN AST`) in the integration suite: every test case, and every ClickHouse keyword as a `WITH` list's name ahead of each statement a `WITH` list can lead. - **A failed ClickHouse query answers by what went wrong, not a flat `500`/`502`** (`internal/api/ch_errors.go` (new, + tests), `internal/api/{errors,query,structured_query,pipes,schema,ch_settings}.go`, `internal/chconn/errclass.go` (`HTTPStatus` exported), `clients/ts/src/errors.ts` (+ tests), `tests/integration/query_errors_test.go` (new), `tests/integration/query_limits_test.go`, `internal/app/app_test.go`, `tests/e2e/sdk/{admin,query}.test.ts`, `AGENTS.md`, `docs/src/content/docs/{api,architecture}.md`, `docs/src/content/docs/{access-control,configuration}.mdx`, `docs/src/content/docs/sdk/{reference.md,index.mdx}`): fixes [#403](https://github.com/Wave-RF/WaveHouse/issues/403) and [#271](https://github.com/Wave-RF/WaveHouse/issues/271), part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). ClickHouse answers a syntax error, a missing grant and an overloaded server alike with HTTP `500`, so `/v1/ops/query` turned a bad statement into a `502` and `/v1/query` and pipes into a `500` the SDK retried. All three now class the failure with `chconn.Classify` through one helper, `writeCHError`: a statement ClickHouse refused is `400 clickhouse.rejected`; a query over a rows/bytes limit, the role's own memory cap, or its time cap where that is no longer than `query_timeout` is `400 clickhouse.limit_exceeded`; `ACCESS_DENIED` is `403 clickhouse.access_denied`; credentials, user or database refused, or a redirect or `4xx` with no exception code from whatever fronts ClickHouse, is `502 clickhouse.misconfigured`; ClickHouse down, unreachable or overloaded is `503 clickhouse.unavailable` with `Retry-After: 5`; a failure with no verdict stays `500` (`502` on the proxy) as `clickhouse.unknown`. The error envelope gains `code` and `retryable` next to `error` on these responses — additive. A role with `max_execution_time` now queries with no context deadline and a cancel two seconds past the cap instead: clickhouse-go overwrote the cap's `max_execution_time` with deadline+5s for any deadline over 1s, so an overrun came back as a bare deadline, indistinguishable from waiting for a pooled connection; ClickHouse now enforces the cap itself and reports `TIMEOUT_EXCEEDED`. `POST /v1/ops/schema/refresh` against an unreachable ClickHouse is a `503` with `Retry-After` instead of a `500`. **SDK:** `WaveHouseError.code` and `retryable` now take the server's `code`/`retryable` when the body has them (`HTTP_` and "5xx retries" otherwise), so a rejected query is `clickhouse.rejected` rather than `HTTP_500`, and is not retried. diff --git a/cmd/wavehouse/main_test.go b/cmd/wavehouse/main_test.go index d0e8404c7..ff471474a 100644 --- a/cmd/wavehouse/main_test.go +++ b/cmd/wavehouse/main_test.go @@ -20,6 +20,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/config" "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" ) // run reads the whole boot config from the environment here (no config @@ -78,7 +79,7 @@ func seedSettings(t *testing.T) string { func TestRun_BootsAndStopsOnCancel(t *testing.T) { hermeticEnv(t) t.Setenv(config.EnvSettingsDir, seedSettings(t)) - t.Setenv("WH_DATA_DIR", t.TempDir()) + t.Setenv("WH_DATA_DIR", storedir.New(t)) _, port, err := net.SplitHostPort(closedAddr(t)) require.NoError(t, err) t.Setenv("WH_SERVER_PORT", port) diff --git a/cmd/wavehouse/validate_test.go b/cmd/wavehouse/validate_test.go index e2e18e6c0..5598cf650 100644 --- a/cmd/wavehouse/validate_test.go +++ b/cmd/wavehouse/validate_test.go @@ -19,7 +19,7 @@ func writeSettingsDir(t *testing.T, policies string) string { "roles.json": `{"roles": ["public"]}`, "policies.json": policies, "pipes.json": `{}`, - "config.json": `{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": 10000, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, + "config.json": `{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": 10000, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, } for name, content := range files { require.NoError(t, os.WriteFile(filepath.Join(dir, name), []byte(content), 0o600)) diff --git a/config.yaml b/config.yaml index 0f0a6ad59..964f37462 100644 --- a/config.yaml +++ b/config.yaml @@ -51,12 +51,22 @@ clickhouse: password: "" max_total_conns: 0 # ceiling on open native connections across pools; 0 = none -# Each layer's implementation, chosen at boot. The in-process backend is the -# default for each, and the only one for these three. +# Each layer's implementation, chosen at boot. The in-process backend is +# each layer's default. mq: backend: embedded # NATS JetStream under /nats dedupe: - backend: pebble # Pebble under /pebble + backend: pebble # Pebble under /pebble; or dynamodb (below) + lease: 30s # how long a claimed id stays pending; at most 59s with the embedded mq (lease + ceil(lease) + 1s within its 2m duplicate window) + reserve_concurrency: 64 # parallel calls per Reserve/Commit/Release to a remote backend, and DynamoDB's idle connections per host; ingest sends a window of up to 256 ids per call + # dynamodb: # read only when backend is dynamodb; credentials from the AWS SDK chain + # table: wavehouse-dedupe-prod + # region: "" # empty = AWS_REGION + # endpoint: "" # dynamodb-local only + # timeout: 250ms + # max_attempts: 3 + # retry_mode: standard # or adaptive + # create_table: false # dynamodb-local only; refused without endpoint coord: backend: local # leases (the sweeper's) held in this process @@ -88,12 +98,12 @@ auth: # clickhouse wiring (addr, http_port, http_scheme, database, username, # query_timeout, tls, headers, max_open_conns, max_idle_conns), auth # (jwks_url, role_claim), dedupe (enabled/id_field/ -# require_id + per-table overrides), dlq.enabled (+ per table), +# require_id/retention + per-table overrides), dlq.enabled (+ per table), # query.default_max_rows / timestamp_bucket_seconds, # schema.refresh_interval, stream keepalive_interval / keepalive_buckets / # gap_window_minutes, mq.max_bytes_gb, cors.allowed_origins — and every key -# is required: the -# binary has no compiled defaults, so what's adopted is exactly what the +# is required except dedupe.retention (missing = "0", forever): the binary +# has no other compiled default, so what's adopted is exactly what the # files say. The server validates the directory at boot (invalid or missing # refuses to start) and reloads it on SIGHUP, on file change, or via # POST /v1/ops/settings/reload; a reload that fails validation keeps the diff --git a/deployments/compose/settings/config.json b/deployments/compose/settings/config.json index 030d76cc1..6b33f55c5 100644 --- a/deployments/compose/settings/config.json +++ b/deployments/compose/settings/config.json @@ -26,6 +26,7 @@ "enabled": false, "id_field": "event_id", "require_id": false, + "retention": "0", "tables": {} }, "dlq": { diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index b50c52525..3e8f6815f 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -285,7 +285,7 @@ The body is a **flat JSON object** whose keys must match column names in the tar | 400 | `{"error":"invalid json"}` | Malformed request body | | 400 | `{"error":"unknown column ... for table ..."}` (also: `missing required column ...`, `type mismatch for column ...`, `null value for non-nullable column ...`) | Schema validation failure (unknown fields, type mismatches, missing required columns, null in a non-nullable column with no default). The body is the validator's message verbatim — there is no `validation failed:` prefix. | | 400 | `{"error":"column \"x\" of table \"t\" is materialized and cannot be inserted"}` (also `… is alias …`) | The record supplies a value for a column ClickHouse computes. Omit it — the server fills it in. Refused rather than dropped: the published row has one slot per insertable column, so the value would otherwise vanish behind a `200` | -| 400 | `{"error":"missing dedupe id field \"event_id\""}` | Only when dedupe is enabled with `dedupe.require_id: true` and the row lacks the configured `id_field`. With `require_id: false` (the default) the row is instead published un-deduped. Either way — reject or publish — the row is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`. In a batch this is a per-record failure, not a whole-request error. | +| 400 | `{"error":"missing dedupe id field \"event_id\""}` | Only when dedupe is enabled with `dedupe.require_id: true` and the row lacks the configured `id_field` or sets it to `null`. With `require_id: false` (the default) the row is instead published un-deduped. Either way — reject or publish — the row is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`. In a batch this is a per-record failure, not a whole-request error. | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (the gate surfaces the token reason rather than silently falling back to `default_role`) | | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table | | 403 | `{"error":"column \"x\" not allowed for insert"}` | The record names a column the role's `allow_columns`/`deny_columns` forbids ([Access control → Column permissions](/access-control#column-permissions)) | @@ -295,10 +295,12 @@ The body is a **flat JSON object** whose keys must match column names in the tar | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, …"}` (declared variant: `Content-Type "text/plain": ingest requires one of …` — see the note above on how declarations are echoed; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | The request declared no `Content-Type`, one whose media type is unsupported or does not parse, a comma-bearing value that does not parse as a single media type, or repeated header lines that disagree — different formats, or one supported and one not. Checked before the body is parsed | | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | +| 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now: it is not open (for example, it failed to open on a reload), or a DynamoDB table is throttling, timing out or unreachable; `Retry-After: 5`. Nothing was published, so the retry is safe | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet (its ClickHouse unreachable, or [no pool for it](/settings-directory#clickhouse)), so whether the table exists is not known; `Retry-After: 5`. Decided before the body is read | -| 500 | `{"error":"publish failed"}` | Message queue error | -| 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure, for that tenant alone) or not open (see [Message Queue](/settings-directory#message-queue)). Response includes `Retry-After: 30` header. | -| 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (a transient broker failure, not a full queue). Response includes `Retry-After: 5` header. Reserved for an external broker ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)): the embedded broker never reports this, and its publish failures are the `500` above. | +| 500 | `{"error":"publish failed"}` | Message queue error whose outcome is unknown, other than a full queue or an unreachable broker (below): the event may have been stored. With dedupe on, the record's id is left to lapse with the dedupe lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default) rather than given back: a retry inside the lease answers the in-flight `503`, and one after it is published under the same idempotency key, which the queue drops if the first copy was stored. The queue's duplicate window (two minutes) covers up to ~2×lease plus a margin, not just the lease itself, so a retry timed off `Retry-After` anywhere in this flow stores no second copy; a much later one is stored again. | +| 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure, for that tenant alone) or not open (see [Message Queue](/settings-directory#message-queue)). Response includes `Retry-After: 30` header. With dedupe on, the record's id is given back, so the retry is published rather than reported as a duplicate. | +| 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`, a transient broker failure, not a full queue) — reserved for an external broker ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)): the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the record's id is left to lapse rather than given back, so a retry cannot land as a second copy; `Retry-After` is that lease, rounded up to whole seconds, when dedupe was on for the record, else the flat `Retry-After: 5`. | +| 503 | `{"error":"a request with the same dedupe id is in flight"}` | Dedupe is on and another request carrying the same id is still being published — usually a client's timeout-retry racing its own original. Its outcome decides whether this record is a duplicate, so retry after the `Retry-After` header (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | **curl example:** @@ -408,13 +410,15 @@ A `200` is returned whenever the body was read and the records were processed | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table (checked once, before any record) | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, …"}` (declared variant: `Content-Type "text/plain": ingest requires one of …` — see the note above on how declarations are echoed; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | The request declared no `Content-Type`, one whose media type is unsupported or does not parse, a comma-bearing value that does not parse as a single media type, or repeated header lines that disagree — different formats, or one supported and one not. Checked before the body is parsed | -| 500 | `{"error":"publish failed"}` / `{"error":"dedupe failed"}` | Message-queue or dedup-backend failure mid-batch | -| 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure) or not open, mid-batch; includes `Retry-After: 30` | -| 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time, mid-batch; includes `Retry-After: 5`. Reserved for an external broker ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)): the embedded broker never reports this, and its publish failures are the `500` above | +| 500 | `{"error":"publish failed"}` / `{"error":"dedupe failed"}` | Message-queue or dedup-backend failure mid-batch, other than a full queue or an unreachable broker (below). After a publish failure the records before it keep their ids, so a whole-batch retry reports those as duplicates; the failing record's id is left to lapse as on the single-object path, and the rest of its window's ids are given back | +| 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure) or not open, mid-batch; includes `Retry-After: 30`. The records before the refused one keep their ids, and its id and the rest of its window's are given back | +| 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`), mid-batch — reserved for an external broker ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)): the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the failing record's id is left to lapse rather than given back — so `Retry-After` is that record's dedupe lease, rounded up to whole seconds, when it was deduped; a record published un-deduped has no lapsing claim to wait out, so `Retry-After: 5` | +| 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now; `Retry-After: 5`. Nothing in the window being reserved was published; the windows before it were, and keep their ids | +| 503 | `{"error":"a request with the same dedupe id is in flight"}` | A record's dedupe id is held by another request still being published; includes `Retry-After` (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). Nothing in that record's window was published; the windows before it were | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] -A batch aborted partway (a `503`/`500`, a JSON-array syntax error, or an NDJSON line over the 10 MiB line bound, after some leading records were already published) re-publishes those leading records when the whole batch is retried. A whole-body read failure is **not** one of these: a `413`, or the `400 invalid request body` of an upload cut off in transit, is decided before any record is processed, so nothing is published — safe to retry, once split for a `413`. Enable deduplication if duplicate suppression matters — this is the same at-least-once property the single-object path already has (the SDK retries both on `503`). +A batch aborted partway (a `503`/`500`, a JSON-array syntax error, or an NDJSON line over the 10 MiB line bound, after some leading records were already published) re-publishes those leading records when the whole batch is retried. Records are published in windows of 256, in order: a read error or a dedupe failure drops the open window unpublished, so what an aborted batch published is the windows before it, plus, after a publish failure, the records of its window before the failing one. A whole-body read failure is **not** one of these: a `413`, or the `400 invalid request body` of an upload cut off in transit, is decided before any record is processed, so nothing is published — safe to retry, once split for a `413`. Enable deduplication if duplicate suppression matters — this is the same at-least-once property the single-object path already has (the SDK retries both on `503`). ::: **curl example (JSON array):** diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 341095dc0..52deec3d2 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -59,10 +59,10 @@ internal/ ├── chsql/ Shared ClickHouse SQL helpers (identifier quoting, bind-safety) ├── config/ YAML + env var configuration loading ├── coord/ Leases for work that must run in one process at a time (the sweeper), with fencing tokens -├── dedupe/ Optional deduplication (Pebble) +├── dedupe/ Optional deduplication (Reserve/Commit/Release; Pebble, DynamoDB) ├── discovery/ ClickHouse schema introspection and validation ├── ingest/ Batch buffering, DLQ, and Active Sweeper -├── keyenc/ The one escaping composite keys are built from (NATS subject tokens, cache keys) +├── keyenc/ The one escaping composite keys are built from (NATS subject tokens, cache keys, dedupe keys) ├── mq/ MQ boundary: the only NATS/JetStream importer (owned message/consumer/stream types + embedded server) ├── observability/ OpenTelemetry pipeline (traces/metrics/logs + Prometheus exposition) ├── pipes/ Named query pipes (NamedQuery type, parameter binding, Source) @@ -83,7 +83,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`clickhouse_exec.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. -- **ingest.go** — Accepts `POST /v1/ingest?table={table}` in three body shapes: one flat JSON object, a JSON array of them, or NDJSON. The **required** `Content-Type` chooses the format *family* — `application/json` versus the four NDJSON spellings — and within the JSON family the body's first non-whitespace byte picks array versus single object; the bytes never choose the family. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type — a comma inside a *quoted* parameter value is data, so `application/json; a=", application/x-ndjson; b="` is accepted. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer and runs the per-format record readers over those bytes, so the `413` lands before any record is processed and peak memory per request is O(body) rather than O(record). Then it validates each record against the discovered schema, optional dedup, and publishes each row through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After: 30`, and a broker that cannot be reached or does not answer in time as `mq.ErrUnavailable`, the `503` + `Retry-After: 5`). When dedup is on, a row missing the configured `id_field` can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). +- **ingest.go** — Accepts `POST /v1/ingest?table={table}` in three body shapes: one flat JSON object, a JSON array of them, or NDJSON. The **required** `Content-Type` chooses the format *family* — `application/json` versus the four NDJSON spellings — and within the JSON family the body's first non-whitespace byte picks array versus single object; the bytes never choose the family. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type — a comma inside a *quoted* parameter value is data, so `application/json; a=", application/x-ndjson; b="` is accepted. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer and runs the per-format record readers over those bytes, so the `413` lands before any record is processed and peak memory per request is O(body) rather than O(record). Then it validates and encodes each record, and runs the records in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row missing the configured `id_field` (or setting it to `null`) can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` (the Go-side type conversion lives in the structured-query / pipes path, not here). - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) driver.Conn` beside it — each resolves the request's tenant per call, and a nil connection (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). @@ -93,7 +93,9 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request ### `app/` — Process wiring - **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability (after which each `Config.Warnings` line is logged at `WARN`), ClickHouse pools, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, the lease coordinator, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. The boot config's `roles` decide which of them a process wires: every process gets the settings registry, observability, the MQ, the coordinator, the reload triggers and a listener; `api` adds schema discovery, the dedupe stores, streaming, auth and the full router; `ingest` adds the ingest worker; `sweeper` adds the sweeper; the ClickHouse pools and the cache come with `api` or `ingest`. A process without `api` serves `api.NewOpsRouter` (probes, `/version`, the metrics path, and the settings reload behind the operator key alone, `wireOpsAuth`) on `server.port`. `config.Validate` refuses a role set the backends cannot serve (a split over the embedded MQ, or `api` without `ingest` and the reverse over a local cache), and `New` refuses a `Config` with no roles, which only one built without `config.Load` can have. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. A layer with a choice of implementation — `wireMQ`, `wireCache`, `wireDedupe`, `wireCoord` — picks it there and nowhere else, in a `switch` on the boot config's `.backend` with one case per backend; the default case refuses boot, which only a `config.Config` built without `config.Load` reaches, since `Validate` refuses a value no case handles. `wireCache` has two: `local`, the in-process `LocalCache`, and `redis`, the shared `RedisCache` built from the `cache.redis` block, which boots bypassed rather than failing when its server is unreachable. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where each would be redelivered every ack wait for as long as the tenant is away and would hold that tenant's ack floor, so the sweeper could purge none of its queue past it. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the whole cache — structured-query and pipe results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultPolicy` reads it through the registry, where a flat directory's tenant `0` is always served. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth, dedupe and cache hooks use too), ending the open streams of a tenant no longer served, and `wireCache`'s hook prunes the cache's version index the same way (`LocalCache.Prune`, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)), so a tenant no longer served stops holding it. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own, and runs only while its process holds the `sweeper` lease (`elected`, which wraps `coord.RunElected` over the coordinator `wireCoord` opens; `local`, the only `coord.backend`, keeps leases in the process, so the one process always holds it). The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe`'s `pebble` case builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ`'s `embedded` case hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. A layer with a choice of implementation — `wireMQ`, `wireCache`, `wireDedupe`, `wireCoord` — picks it there and nowhere else, in a `switch` on the boot config's `.backend` with one case per backend; the default case refuses boot, which only a `config.Config` built without `config.Load` reaches, since `Validate` refuses a value no case handles. `wireCache` has two: `local`, the in-process `LocalCache`, and `redis`, the shared `RedisCache` built from the `cache.redis` block, which boots bypassed rather than failing when its server is unreachable. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where each would be redelivered every ack wait for as long as the tenant is away and would hold that tenant's ack floor, so the sweeper could purge none of its queue past it. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the whole cache — structured-query and pipe results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultPolicy` reads it through the registry, where a flat directory's tenant `0` is always served. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth, dedupe and cache hooks use too), ending the open streams of a tenant no longer served, and `wireCache`'s hook prunes the cache's version index the same way (`LocalCache.Prune`, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)), so a tenant no longer served stops holding it. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own, and runs only while its process holds the `sweeper` lease (`elected`, which wraps `coord.RunElected` over the coordinator `wireCoord` opens; `local`, the only `coord.backend`, keeps leases in the process, so the one process always holds it). The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe`'s `pebble` case builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) and then its table — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. `wireDedupe`'s `dynamodb` case is `wireDynamoDedupe`, in wire_dynamodb.go (below). `wireHTTP` hands the ingest handler `dedupe.lease` (`IngestHandler.DedupeLease`) whichever backend is chosen. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ`'s `embedded` case hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. + +- **wire_dynamodb.go** — `wireDedupe`'s `dynamodb` case, split out of wire.go so the e2e suite's coverage exclude for it (the e2e binary always runs Pebble dedupe, never DynamoDB) doesn't have to blanket wire.go itself: builds the same `dedupe.Stores` over `Dynamo.Tenant`, gated (`Factory.Gated`) on the table's check: boot runs `Dynamo.Check` (after `CreateTable`, when `dedupe.dynamodb.create_table` is on) whether or not any tenant has dedupe on. Boot is refused only for a misconfigured table (an error that is not `ErrUnavailable`) over a flat directory whose tenant has dedupe on; every other failure boots with the switched-on stores closed, the check retried until it passes by a background component that backs off from one second to thirty (a nested directory has no watcher, and a flat one's table can come good with no settings change). The `AfterAdopt` hook never runs the check, since it holds the lock that serializes reloads, and it does not wait on a tenant whose `dedupe.enabled` is unchanged either — `Managed.Apply`'s no-op fast path settles that case under its own read lock, so the hook only takes a store's write lock, and so waits for that tenant's in-flight `Reserve`/`Commit`/`Release` calls to finish, on a genuine flip. It applies every store against the last check's result, so a tenant a reload switches on fails closed meanwhile, and wakes the retry, so a reload still retries at once. It has no Pebble gauges. ### `stream/` — SSE keepalive & fan-out @@ -126,7 +128,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ - **config.go** — Loads *boot* configuration from a YAML file with environment variable overrides (using [cleanenv](https://github.com/ilyakaznacheev/cleanenv)); every key has a `WH_`-prefixed env var. Boot config is only what can't change under a running process — the implementation each layer runs on, the process's `roles`, resource sizing, listeners, observability exporters, the settings-directory path, and the secrets (`clickhouse.password`, `cache.redis.password`, `auth.jwt_secret`, `auth.operator_key`). Everything tenant-tunable lives in the settings directory (`settings/`). Both sources are strict: `Load` refuses to boot naming every YAML key the struct doesn't declare (`strict.go`) and every `WH_*` environment variable no field binds (`check.go`), so a tunable that moved to the settings directory can't be read, ignored, and believed. Boot is the validator for this half — there is no dry-run command. See [Configuration Reference](/configuration). - **check.go** — `rejectUnboundEnv` is the environment half of the strict loader: `unboundEnv` walks the struct's `env` tags (plus the two process-level names, `WH_CONFIG` and `WH_LOG_LEVEL`) against the environment; `CheckDataDir` probes `data_dir` — run by `main` right after `Load` when `Config.NeedsDataDir` says a selected backend keeps state there, so an unusable `data_dir` refuses boot before anything dials out. It refuses an empty value (reachable through `WH_DATA_DIR=`) outright rather than probing the working directory; a path that exists and is not a directory; a dangling symlink at `data_dir` or any component above it (the walk to the nearest existing ancestor uses `Lstat`, so a failed mount is not skipped over as "does not exist"); and a directory the process cannot write to — or, when it does not exist, an unwritable nearest ancestor — probed by creating and removing one temp file. A permission denial, on the probe or on reaching the path through a parent without search permission, carries the UID-65532 hint, since a bind mount owned by root is the typical cause. -- **backends.go** — the `.backend` keys: one string type per layer (`MQBackend`, `CacheBackend`, `DedupeBackend`, `CoordBackend`), each with its list of the backends this build has, and a `validate` per layer block (`checkBackend`, run by `validateBackends` in `Validate`, between `validateRoles` and `validateTopology`), which refuses a value not on the list and names the ones that are. A backend's own settings go in a `.` sub-block that is its `validate`'s case to check. `Distributed` reports whether the MQ is shared with other processes, `NeedsDataDir` whether a selected backend keeps state under `data_dir`, and `Warnings` returns what a valid configuration is still likely to get wrong — the combinations that are correct for one replica only (a shared MQ over a local cache or Pebble dedupe), a `cache.redis` block that is not read, certificate verification turned off — which `app.New` logs at `WARN`. +- **backends.go** — the `.backend` keys: one string type per layer (`MQBackend`, `CacheBackend`, `DedupeBackend`, `CoordBackend`), each with its list of the backends this build has, and a `validate` per layer block (`checkBackend`, run by `validateBackends` in `Validate`, between `validateRoles` and `validateTopology`), which refuses a value not on the list and names the ones that are. One rule spans two layers: while `mq.backend` is `embedded`, `dedupe.lease` plus its own ceiling to the next whole second (`ceilSecond`) plus one more second must fit the embedded MQ's 2m duplicate window (`embeddedDuplicateWindow`), a cap of 59s (`maxEmbeddedLease`), because a client obeying the in-flight `503`'s `Retry-After` can republish as late as the lease plus that ceiling plus a second after the claim, since DynamoDB rounds a claim's expiry up to the second. A backend's own settings go in a `.` sub-block that is its `validate`'s case to check. `Distributed` reports whether the MQ is shared with other processes, `NeedsDataDir` whether a selected backend keeps state under `data_dir`, and `Warnings` returns what a valid configuration is still likely to get wrong — the combinations that are correct for one replica only (a shared MQ over a local cache or Pebble dedupe), a `cache.redis` block that is not read, certificate verification turned off — which `app.New` logs at `WARN`. - **cache_redis.go** — `CacheRedisConfig`, the `cache.redis` sub-block, and its checks: an address (exactly one in `standalone` mode, which dials only the first), each `host:port` with a port from 1 to 65535 (a URL or `user:password@` form refused without repeating it, since it may hold a password), a known mode (`sentinel` is refused until [#656](https://github.com/Wave-RF/WaveHouse/issues/656)), `db` 0 in cluster mode, positive timeouts and sizes, a `timeout` and `dial_timeout` of at most 1 s each (boot and `Close` each wait out a dial: a connect and a handshake bounded by `dial_timeout`, and a cluster's topology read bounded by the larger of the two), a `version_ttl` of at least 2 s, and a `compress_min_bytes` that is not negative (`0` never compresses); its defaults are in `defaults()` with the rest. `CacheRedisTLS.Config` builds the `tls.Config`, reading the files; `Validate` calls it so an unreadable file refuses boot, and `internal/app` calls it again to build the connection. A TLS key set while `tls.enabled` is off is an error rather than a plaintext connection. - **config.go**, roles — `roles` (`[]Role`: `api`, `ingest`, `sweeper`; `AllRoles` by default; `Has(Role)`) picks which components `internal/app` wires, and `instance_id` names the process (`-<8 hex>` when empty, resolved in `Load`; today only logged at boot, and a distributed coordinator will record it as a lease's holder). `validateRoles` refuses an empty list, an empty entry, an unknown or a repeated role; `validateTopology` refuses a role set the backends cannot serve: any split over the embedded MQ, and a process with exactly one of `api` and `ingest` over a local cache. `NeedsDataDir` counts Pebble only for a process running `api`, and `Warnings` is empty without `api`, since only that role opens a cache it reads or a dedupe store. - **strict.go** — `rejectUnknownKeys`, the YAML half: re-reads the file as a generic tree and walks it against the struct's `yaml` tags, listing every key the struct doesn't declare. cleanenv itself is lenient by design, which is exactly wrong for boot config once keys have moved to the settings directory. @@ -141,10 +143,13 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `dedupe/` — Deduplication (Optional) -- **dedupe.go** — `Deduplicator` interface: `CheckAndMark(ctx, eventID) (bool, error)`. -- **embedded.go** — `Embedded`, the [Pebble](https://github.com/cockroachdb/pebble) (embedded key-value store) implementation: every tenant's seen ids in one instance at `data_dir/pebble` ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 3), key = tenant id, a NUL, event id — no tenant id holds a NUL, so no two tenants' keys meet. `NewEmbedded(dataDir)` opens nothing; `Tenant(id)` is the `Factory` a `Stores` takes, building the tenant's `Managed` over its share of the instance, which opens with the first tenant store switched on and closes with the last one switched off. `Stats` reports the instance's figures for the system gauges, nil while it is closed. -- **managed.go** — `Managed` wraps one store — opened through the function `NewManaged` takes, so the switch semantics are the same for every backend — behind the hot-reloadable `dedupe.enabled` switch: `Apply(enabled)` opens or closes it, idempotently, and in-flight `CheckAndMark` calls are serialized against the swap, so flipping the key is a reload, not a restart. `CheckAndMark` returns `ErrDisabled` while switched off (the ingest handler publishes un-deduped and counts it — a reload-window race, not a mode) and `ErrUnavailable` while switched on but not open (ingest fails closed). -- **stores.go** — `Stores` is one `Managed` per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7), built on first use through a `Factory` (`func(tenant.ID) *Managed`) — whether tenants share a backend is the factory's business (`Embedded.Tenant` puts them all in one Pebble instance), with nothing that holds the `Stores` changing. `For(id)` returns a tenant's store, built closed so a tenant adopted a moment ago answers `ErrDisabled` rather than having no store; `Retain(keep)` closes and forgets the stores of tenants no longer served, touching nothing on disk; `Close()` closes every store. `internal/app` drives it from the registry's `AfterAdopt` hook. +- **dedupe.go** — the `Deduplicator` contract, two-phase: `Reserve(ctx, keys, lease)` answers one `Claim` per `Key{Table, ID}`, in order — `Claimed` (first sighting: the caller now holds a pending claim), `Duplicate` (committed earlier, or repeated earlier in the same call) or `InFlight` (another request holds a live claim) — and is atomic per key across every process sharing the backend; `Commit(ctx, claims, retention)` makes the published ids duplicates (retention `0` = forever); `Release(ctx, claims)` gives back ids whose records were definitely not published (a refused or never-sent publish; one whose outcome is unknown is left to lapse instead). A claim neither committed nor released lapses after its lease, so a request that dies mid-publish never strands an id. There is deliberately no read-only check: a separate read is how [#390](https://github.com/Wave-RF/WaveHouse/issues/390) happened. +- **key.go** — the key every backend stores, as text: `/
/` (for example `acme/clicks/evt-123`, [#222](https://github.com/Wave-RF/WaveHouse/issues/222)). The table and id are escaped and joined (`keyenc.AppendJoin`) by `internal/keyenc` — the escaping NATS subject tokens use — which never writes `/`, and a tenant id cannot hold one, so a table name may hold any byte, NUL included, and neither tenants nor tables share ids; a key is ASCII, so it reads as-is in a console and is a valid DynamoDB String. An id whose escaped form is over 1,024 bytes is stored as `#` plus its SHA-256 in hex (`#` is never written by the escaping), counted by `wavehouse_dedupe_hashed_id_total`. +- **embedded.go** — `Embedded`, the [Pebble](https://github.com/cockroachdb/pebble) (embedded key-value store) implementation: every tenant's seen ids in one instance at `data_dir/pebble` ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 3). Pending claims live in memory beside it, in 64 locked shards: one process owns the instance, so a crash forgetting them is every lease lapsing at once, and the shard lock makes check-and-claim atomic. `Commit` writes every claim it is given in one batch and one fsync, each value carrying its expiry (`0` = never), which `Reserve` honors on read. A background sweep (`sweep.go`), started when the instance opens and stopped before it closes, deletes expired keys and the version-0 keys from before the table joined the key — told apart by their value, which is never a current commit's, since a bare id from before tenants led the key could spell a current one ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)): a minute after opening, then hourly, 1,024 keys per chunk, read without a lock, so the deleted keys a chunk steps over (Pebble keeps them until it compacts) never hold up a `Commit`, then re-read under a lock `Commit` also takes and deleted only if still expired or version-0, so a key re-committed after the sweep read it is never deleted; `wavehouse_dedupe_swept_keys_total{reason}` counts what it deletes. `NewEmbedded(dataDir)` opens nothing; `Tenant(id)` is the `Factory` a `Stores` takes, building the tenant's `Managed` over its share of the instance, which opens with the first tenant store switched on and closes with the last one switched off. `Stats` reports the instance's figures for the system gauges, nil while it is closed. Pebble is per process: two pods on it do not share seen ids. +- **dynamodb.go** — `Dynamo`, the DynamoDB implementation, selected by `dedupe.backend: dynamodb`: every tenant's ids in one shared table, `pk` (a string) the key above, no sort key. `Reserve` is a conditional `PutItem` per key, run in parallel up to `ReserveConcurrency` (64), with at least as many idle connections kept per host so a wide `Reserve` reuses them rather than dial. It succeeds when no live item holds the key, where an item whose `ex` (epoch seconds, rounded up) has passed counts as absent whether or not TTL has deleted it yet. On a failed condition, the returned old item says `Duplicate` or `InFlight` without a read, or `Claimed` when it is the put's own pending item (same token): an SDK retry of an attempt DynamoDB applied but whose answer was lost. If any put errors, or the caller cancels (a client disconnecting mid-request), the puts not yet sent are skipped and every put that may have landed is released by its token. A put already sent runs to its answer on a context the caller's cancellation does not reach, so it answers before that undo; only its own call deadline can cut it off, and DynamoDB may then apply it after its release, holding its key `InFlight` until the lease ends, as a crashed request's claim does. `Commit` is `BatchWriteItem`, 25 at a time, retrying with jittered backoff, for up to eight rounds, both the items DynamoDB leaves unprocessed and a batch that failed transiently (a throttle means it processed none of it); the records are already published, and a table that throttles every round delays the ingest response by at most about 3 s at the defaults (eight 250 ms calls and the waits between them) before the commit is given up. `Release` is a `DeleteItem` conditional on the token and the pending state. Each call has a `Timeout` (250 ms) covering the SDK's retries (`MaxAttempts`, 3), which back off with full jitter under a ceiling capped at `Timeout/(2·(MaxAttempts−1))`, so a call's retries wait at most half its timeout and a throttled call fails on its last attempt's answer rather than on the deadline. Throttling, server faults, timeouts and connection failures wrap `ErrUnavailable`; a missing table or denied access does not, since those are configuration bugs. Five unavailable claims in a row within a second short-circuit `Reserve` for a second, for every tenant (one breaker per `Dynamo`). `NewDynamo` builds the client from the AWS SDK's default chain (Pod Identity or IRSA), with an `Endpoint` override for dynamodb-local, and refuses a config that resolves no region. `Check` verifies the key schema and warns when TTL is off. `CreateTable` is refused unless `Endpoint` is set. `Tenant(id)` is the `Factory`. The table definition and IAM policy are on the [Deployment](/deployment) page. +- **managed.go** — `Managed` wraps one store — opened through the function `NewManaged` takes, so the switch semantics are the same for every backend — behind the hot-reloadable `dedupe.enabled` switch: `Apply(enabled)` opens or closes it, idempotently, and in-flight calls are serialized against the swap, so flipping the key is a reload, not a restart. Every call returns `ErrDisabled` while switched off (the ingest handler publishes un-deduped and counts it — a reload-window race, not a mode) and `ErrUnavailable` while switched on but not open (ingest fails closed). `Reserve` also reads a lease `<= 0` as `DefaultLease` and collapses a key repeated in one call before the backend sees it, once for every backend — so a backend may assume distinct keys, a positive lease, and only `Claimed` claims in `Commit` and `Release`. +- **dedupetest/** — the conformance suite every backend runs: `Run(t, newHarness)` drives the contract above through a backend's `Factory` (claim, commit, release, lease lapse, retention, one claim among concurrent reserves from two clients, keyspaces, input order, late commit, stale release, failure mid-call); `Harness` optionally injects a clock and a mid-call failure. `Mark` is the old check-and-mark in one call, for tests that only need an id seen. +- **stores.go** — `Stores` is one `Managed` per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7), built on first use through a `Factory` (`func(tenant.ID) *Managed`) — whether tenants share a backend is the factory's business (`Embedded.Tenant` puts them all in one Pebble instance), with nothing that holds the `Stores` changing. `For(id)` returns a tenant's store, built closed so a tenant adopted a moment ago answers `ErrDisabled` rather than having no store; `Retain(keep)` closes and forgets the stores of tenants no longer served, touching nothing on disk; `Close()` closes every store. `Factory.Gated(ready)` wraps a factory so a store opens only once `ready` returns nil, and fails closed until then (the DynamoDB wiring's table check). `internal/app` drives it from the registry's `AfterAdopt` hook. ### `discovery/` — Schema Discovery & Validation @@ -165,11 +170,11 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ The **only** package that imports NATS/JetStream — a `depguard` rule in `.golangci.yml` fails `make lint` on any `github.com/nats-io` import in every package golangci-lint builds; the `integration`-tagged files under `tests/` sit outside its default build context, so the boundary there rests on convention (AGENTS.md Key Design Decision #20). Every other package talks to the broker through the types below, so a subject, stream, or broker change lands here once. -- **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`, `NakWithDelay(d)`, which falls back to `Nak` for a message built without `WithNakDelay`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After: 30`; `ErrUnavailable` when the broker cannot be reached or does not answer in time — the 503 + `Retry-After: 5`, which no backend returns yet: the embedded broker's publish failures are the `500`), `Subscriber` (every ingest event of every tenant, under a named durable consumer, its fetch-ahead split across the tenants — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; one error per failed tenant, joined — `ErrConsumerNotFound` for a queue the consumer has not been created on yet, the one failure the sweeper logs as a warning rather than an error) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. +- **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`, `NakWithDelay(d)`, which falls back to `Nak` for a message built without `WithNakDelay`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape; `WithIdempotencyKey` marks a publish so that a second one carrying the same key inside the queue's duplicate window is dropped and reported as success. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After: 30`; `ErrUnavailable` when the broker cannot be reached or does not answer in time — the 503 + `Retry-After: 5`, which no backend returns yet: the embedded broker's publish failures are the `500`), `Subscriber` (every ingest event of every tenant, under a named durable consumer, its fetch-ahead split across the tenants — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; one error per failed tenant, joined — `ErrConsumerNotFound` for a queue the consumer has not been created on yet, the one failure the sweeper logs as a warning rather than an error) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. - **subject.go** — The embedded broker's naming, private to the package: the stream names (`INGEST_` and `DLQ_` — prefixes that differ in their first letter, so no tenant id makes one kind's name the other's — and the one pair an earlier build kept for every tenant together, `WAVEHOUSE`/`WAVEHOUSE_DLQ`, which boot deletes), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject tokens (`internal/keyenc`: ASCII letters, digits, `_` and `-` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded — and the tail's first token picks the tenant's stream. - **deadletter.go** — `deadLetterTables`, the per-table count `DeadLetterCounts` reports: a dead-letter stream's per-subject counts, each subject parsed back to its topic and counted under its table — every scope of a table under the table itself, so a dotted table name never shares a count with a table + scope pair — and a table filter keeps that table with all of its scopes. - **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts that tenant's purge. It runs on each tenant's stream at that tenant's cutoff, and a failure on one tenant's stream is reported without stopping the sweep of the others. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. -- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, and so does a publish to a queue the broker has not recorded open — an open that timed out can leave a stream JetStream creates after all, which no consumer holds — either refused as a full queue with none asked yet. Publishes and parks that find the same queue not open share one attempt (`singleflight`), and after one fails the tenant's publishes and parks are refused at once for five seconds rather than each trying again under the broker's lock, which every tenant's open, resize and reload takes; a reload retries regardless. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. +- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`) and remembering idempotency keys for `EmbeddedDuplicateWindow` (two minutes, sized to `2 × the dedupe lease + 1s` — the in-flight `503` sends the full lease as `Retry-After`, so an obedient client's retry can land up to ~2×lease after the original `Reserve`, and the `+1s` covers a backend whose claim expiry itself rounds up by that much), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, and so does a publish to a queue the broker has not recorded open — an open that timed out can leave a stream JetStream creates after all, which no consumer holds — either refused as a full queue with none asked yet. Publishes and parks that find the same queue not open share one attempt (`singleflight`), and after one fails the tenant's publishes and parks are refused at once for five seconds rather than each trying again under the broker's lock, which every tenant's open, resize and reload takes; a reload retries regardless. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. - **mqtest/** — The conformance suite for `Broker` (`mqtest.Run`): the behavior the rest of the process relies on — publish and consume round trips with names that need encoding, per-tenant order, redelivery, dead-lettering and its counts, replay bounds and isolation, the one `failed` report of a consumer whose delivery ends underneath it — checked through the interfaces alone, with no stream or subject name in sight. Each implementation runs it from a test of its own — the embedded one from `mqtest/embedded_test.go`, a test binary apart from `internal/mq`'s so the two share no 15s budget — handing it a fresh broker per case and flags (`mqtest.Caps`) for the few places where backends legitimately differ: whether a full queue refuses its own tenant alone, whether `PurgeAcked` removes anything, whether a tenant never given a budget has a dead-letter queue to report on, and whether `CreateConsumer` configures the durable or only finds one. ### `observability/` — OpenTelemetry Pipeline @@ -208,7 +213,7 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi - **store.go** — `Store` is a passive holder: one tenant's adopted document behind an atomic pointer, swapped by the registry, which stamps it with the id of the tenant it created the store for (`Tenant()`, how a handler names its tenant to a per-tenant resource). Consumers read typed accessors per call (`ClickHouse()`, `Auth()`, `DedupeFor(table)`, `DLQFor(table)`, `Keepalive()`, …) rather than holding values. - **registry.go** — `Registry` maps a tenant id to its `Store` and owns everything that changes one. `Open` validates and adopts at boot; `Reload` re-validates the whole directory and `ReloadTenant` one tenant's folder, serialized with each other; `AfterAdopt` hooks run after every reload the registry applied, with the tenants it adopted — none when it only rejected or removed one, which a consumer holding a resource per tenant needs to hear of too; `For(id)` and `All()` see only the tenants being served, `Known()` every tenant held, rejected ones included, and `Resolve(id)` tells a rejected tenant from an unknown one. The shape is fixed at `Open`. Flat: an invalid directory refuses boot, and a rejected reload keeps the previous snapshot. Nested: fail closed per tenant — a folder with an error finding stops being served (the store keeps its document for requests already admitted, and gets the next good one) while the rest carry on; a whole-directory reload mirrors the folders, down to none (an emptied directory is not a change of shape); and a finding about the directory itself refuses boot or rejects the reload whole, leaving every tenant as it was. The tenant map is replaced whole by a reload, so a lookup is one lock-free load. - **watch.go** — `Registry.Watch`, which `internal/app` starts for a flat directory only: fsnotify on the *directory* (not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost), debounced into one reload; reloads once as soon as the watch exists so an edit between the boot read and the watch is never missed. `SIGHUP` and the reload endpoint funnel through the same serialized `Reload`. -- **seed.go** / **seed/** — The embedded (`go:embed`) starter directory with every key at its default. The binary carries no compiled defaults: `wavehouse bootstrap [dir]` writes this seed, and the compose stack and e2e fixture ship copies of it. +- **seed.go** / **seed/** — The embedded (`go:embed`) starter directory with every key at its default. The binary carries no compiled defaults except that a missing `dedupe.retention` means `"0"`: `wavehouse bootstrap [dir]` writes this seed, and the compose stack and e2e fixture ship copies of it. ### `tenant/` — Tenant Identifier @@ -225,7 +230,7 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi ### `keyenc/` — Key Escaping -- **keyenc.go** — The one escaping composite keys are built from, so a name can never be mistaken for a separator: `Escape` keeps ASCII letters, digits, `_` and `-` — exactly the tenant-id grammar, so a tenant id is its own escaped form — and writes every other byte as `%XX` (uppercase hex); `Unescape` is `url.PathUnescape`, which decodes `%XX` in either case and takes any other byte as itself, so a `%2D` for `-` that an earlier build wrote still reads. `Join`/`AppendJoin` escape each field and put a separator between them, panicking on no fields and on a separator the escaping could write or one outside ASCII, and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to: NATS subject tokens (`internal/mq`) and the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — use it. Keys built from it are stored, so changing what it keeps orphans them — and on the shared backend, whose keys every process builds for itself, it splits them for the length of a rolling upgrade: a bump one build makes does not reach the entries the other build filed, which are served until their TTL. +- **keyenc.go** — The one escaping composite keys are built from, so a name can never be mistaken for a separator: `Escape` keeps ASCII letters, digits, `_` and `-` — exactly the tenant-id grammar, so a tenant id is its own escaped form — and writes every other byte as `%XX` (uppercase hex); `Unescape` is `url.PathUnescape`, which decodes `%XX` in either case and takes any other byte as itself, so a `%2D` for `-` that an earlier build wrote still reads. `Join`/`AppendJoin` escape each field and put a separator between them, panicking on no fields and on a separator the escaping could write or one outside ASCII, and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to: NATS subject tokens (`internal/mq`), the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — and the dedupe keys (`internal/dedupe`, `/`-separated) use it. Keys built from it are stored, so changing what it keeps orphans them — and on the shared backend, whose keys every process builds for itself, it splits them for the length of a rolling upgrade: a bump one build makes does not reach the entries the other build filed, which are served until their TTL. ## Data Flows @@ -251,11 +256,21 @@ Client POST /v1/ingest?table={table} → Canonicalize top-level DateTime/DateTime64 column values to RFC 3339 UTC (rewrites the payload so every consumer shares one spelling; fail-open — an unparseable value passes through verbatim for ClickHouse's parser to judge) - → Optional deduplication check (configurable ID field; a row missing that - field is published un-deduped + logged/counted, or rejected under require_id) - → Publish to NATS JetStream (ingest.{tenant}.{table}) + → Optional dedupe: resolve the id (configurable ID field; a row missing it or + setting it to null is published un-deduped + logged/counted, or rejected + under require_id) + → Encode the record; the steps below run per window of up to 256 records + → Reserve the window's (tenant, table, id) keys in one call: a duplicate is + skipped, an id another request holds → 503 + Retry-After (dedupe.lease, + 30s by default), a store that cannot answer → 503 + Retry-After: 5 + → Publish each record to NATS JetStream (ingest.{tenant}.{table}), a deduped + one under its idempotency key + → Commit the published ids in one call; on a failed publish, commit the + records before it and release the rest (a publish whose outcome is unknown + keeps its claim until the lease lapses) → 200 OK returned immediately - → (If the tenant's NATS stream is full, or not open: 503 + Retry-After header) + → (If the tenant's NATS stream is full, or not open: 503 + Retry-After header, + the id released) Ingest worker pipeline (StartIngestWorker): ← JetStream pull consumer (buffer-consumer), one durable per tenant stream diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index 2bea00fab..8a89f6444 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -35,20 +35,43 @@ This page is boot config only — what the platform operator owns (wiring, lifec | YAML Key | Env Var | Default | Description | | --- | --- | ------- | ----------- | -| `data_dir` | `WH_DATA_DIR` | `./data` | Root directory for embedded state. NATS JetStream lives at `/nats`; Pebble, holding every tenant's dedupe store while any tenant has dedupe enabled, at `/pebble`. Subdirectory names are conventions, not config — one knob, one mount. **In a container this MUST resolve to a host-backed volume**; the relative default is for local binary use. WaveHouse logs a startup `WARN` when the directory is missing or empty (no prior state). See [Persistent Storage](/deployment#persistent-storage-required-for-containers). | +| `data_dir` | `WH_DATA_DIR` | `./data` | Root directory for embedded state. NATS JetStream lives at `/nats`; Pebble (with `dedupe.backend: pebble`), holding every tenant's dedupe store while any tenant has dedupe enabled, at `/pebble`. Subdirectory names are conventions, not config — one knob, one mount. **In a container this MUST resolve to a host-backed volume**; the relative default is for local binary use. WaveHouse logs a startup `WARN` when the directory is missing or empty (no prior state). See [Persistent Storage](/deployment#persistent-storage-required-for-containers). | ### Backends -Each layer's implementation is chosen once, at boot. Every layer defaults to its in-process backend, so a config that sets none of these keys runs as it always has; the cache also has a shared one, `redis`. A value this build has no backend for refuses boot and names the valid ones. +Each layer's implementation is chosen once, at boot. Every layer defaults to its in-process backend, so a config that sets none of these keys runs as it always has; the cache and dedupe also have a shared one each, `redis` and `dynamodb`. A value this build has no backend for refuses boot and names the valid ones. | YAML Key | Env Var | Default | Description | | --- | --- | ------- | ----------- | | `mq.backend` | `WH_MQ_BACKEND` | `embedded` | The message queue. `embedded`: NATS JetStream inside this process, under `/nats`. It listens on no port, so no other process can reach its queue. | | `cache.backend` | `WH_CACHE_BACKEND` | `local` | The query-result cache. `local`: in this process, sized by `cache.l1_max_cost`. `redis`: one Redis-compatible server shared by every process, configured by [`cache.redis`](#cache), so an insert one process makes invalidates what every process has cached. | -| `dedupe.backend` | `WH_DEDUPE_BACKEND` | `pebble` | Where ingest dedupe keeps the event ids it has seen. `pebble`: in this process, under `/pebble`, open while any tenant has dedupe on. | +| `dedupe.backend` | `WH_DEDUPE_BACKEND` | `pebble` | Where ingest dedupe keeps the event ids it has seen. `pebble`: in this process, under `/pebble`, open while any tenant has dedupe on; two processes do not share seen ids. `dynamodb`: one DynamoDB table that every tenant and every process shares, configured by [`dedupe.dynamodb`](#dynamodb-dedupe). | | `coord.backend` | `WH_COORD_BACKEND` | `local` | Where the leases for work only one process may do at a time, such as the sweeper, are held. `local`: in this process, so the one process always holds them. It shares nothing with another process, so every process runs its own sweeper. | -Settings for one backend go in a sub-block named after it, `.`, read only when that backend is selected. `cache.redis` is the only one so far; any other, `mq.embedded` included, is an unknown key and refuses boot. A `cache.redis.addrs` set while `cache.backend` is `local` is logged at `WARN` at boot, since the block is not read. `mq` and `dedupe` also appear in the settings directory's `config.json`, with different keys (`mq.max_bytes_gb`, `dedupe.enabled`, …): those are per-tenant tunables and stay there, and one written in `config.yaml` refuses boot as an unknown key. +Settings for one backend go in a sub-block named after it, `.`, read only when that backend is selected. `cache.redis` and `dedupe.dynamodb` are the only ones so far; any other, `mq.embedded` included, is an unknown key and refuses boot. A `cache.redis.addrs` set while `cache.backend` is `local` is logged at `WARN` at boot, since the block is not read. `mq` and `dedupe` also appear in the settings directory's `config.json`, with different keys (`mq.max_bytes_gb`, `dedupe.enabled`, …): those are per-tenant tunables and stay there, and one written in `config.yaml` refuses boot as an unknown key. + +### Dedupe + +Whether a tenant dedupes, and on which field, are settings-directory keys ([Deduplication](/settings-directory#deduplication)). What is boot config is where the seen ids live and how a claim behaves. + +| YAML Key | Env Var | Default | Description | +| --- | --- | ------- | ----------- | +| `dedupe.lease` | `WH_DEDUPE_LEASE` | `30s` | How long a record's id stays claimed while the record is published. Another request carrying the same id meanwhile gets `503` with this as `Retry-After`, in whole seconds; a claim that is neither committed nor released, because its process died mid-publish, lapses after it. With `mq.backend: embedded`, the lease plus its own ceiling to the next whole second plus one more second must fit the embedded queue's 2-minute duplicate window, so the lease is at most `59s`: a client that obeys `Retry-After` after a publish whose outcome it never learned can republish as late as the lease plus that ceiling plus a second after the claim, since DynamoDB rounds a claim's expiry up to the second. A longer lease refuses boot. A Go duration (`30s`, `45s`); `0` refuses boot. | +| `dedupe.reserve_concurrency` | `WH_DEDUPE_RESERVE_CONCURRENCY` | `64` | The most parallel calls one Reserve, Commit or Release makes to a remote dedupe backend, and the idle connections per host the DynamoDB client keeps to match, never fewer than the SDK's own default (10). Ingest reserves and commits a window of up to 256 ids per call; `pebble` ignores it. `0` refuses boot. | + +#### DynamoDB dedupe + +Read only when `dedupe.backend` is `dynamodb`. Credentials come from the AWS SDK's default chain (EKS Pod Identity or IRSA in a pod; `AWS_*` variables or a profile locally), never from this file; the table and its IAM policy are described in [Deployment](/deployment#a-shared-dedupe-table-on-dynamodb). At boot WaveHouse checks the table: its key schema must be `pk` (String) alone, and TTL off on `ex` is logged as a warning. A misconfigured table — missing, with the wrong key schema, or denied to the process's credentials — refuses boot only with a flat settings directory whose tenant has dedupe on, and is logged at `ERROR` otherwise. In every other case — a transient failure (a throttle, a timeout, the network), a nested directory, or no tenant with dedupe on yet — the process boots, every tenant with dedupe on answers ingest `503` (`dedupe store unavailable`, `Retry-After: 5`) until the check passes, and the check is retried in the background, backing off from one second to thirty, and at once after every reload, so a table that comes good is picked up without a restart. A reload makes no table call, and does not wait on a tenant whose dedupe setting is unchanged: it applies each tenant's switch against the last check's result, so a tenant it switches on fails closed until the retry passes. A reload waits only for a tenant whose store it closes — dedupe switched off, or the tenant removed or rejected — and then only for that tenant's in-flight `Reserve`/`Commit`/`Release` calls, before the store itself closes. The check runs in every process running the `api` [role](#process-roles), the one that opens the dedupe stores, whether or not any tenant has dedupe on. + +| YAML Key | Env Var | Default | Description | +| --- | --- | ------- | ----------- | +| `dedupe.dynamodb.table` | `WH_DEDUPE_DYNAMODB_TABLE` | *(required)* | The shared table. | +| `dedupe.dynamodb.region` | `WH_DEDUPE_DYNAMODB_REGION` | *(empty)* | The table's region. Empty uses the SDK chain's (`AWS_REGION`); no region from either refuses boot. | +| `dedupe.dynamodb.endpoint` | `WH_DEDUPE_DYNAMODB_ENDPOINT` | *(empty)* | A custom endpoint, for [dynamodb-local](https://docs.aws.amazon.com/amazondynamodb/latest/developerguide/DynamoDBLocal.html) in development and tests. Leave it empty against AWS. | +| `dedupe.dynamodb.timeout` | `WH_DEDUPE_DYNAMODB_TIMEOUT` | `250ms` | Deadline for each DynamoDB call, the SDK's retries included. The retries back off with full jitter, each wait capped at `timeout / (2 × (max_attempts − 1))`, so together they wait at most half of it and a throttled call fails on its last attempt's answer rather than on the deadline. The boot and background table check (verifying the key schema and TTL) is not one of these calls: it runs under its own deadline of 10 × `timeout` (`2.5s` by default). `0` refuses boot. | +| `dedupe.dynamodb.max_attempts` | `WH_DEDUPE_DYNAMODB_MAX_ATTEMPTS` | `3` | Attempts per call, the first included. More attempts share the same half of `timeout` for their waits, so each retry waits less rather than the call running longer. `0` refuses boot. | +| `dedupe.dynamodb.retry_mode` | `WH_DEDUPE_DYNAMODB_RETRY_MODE` | `standard` | `standard`, or `adaptive`, which also slows the client down after throttling. Anything else, empty included, refuses boot. | +| `dedupe.dynamodb.create_table` | `WH_DEDUPE_DYNAMODB_CREATE_TABLE` | `false` | Development only: create the table at boot if it is missing, with TTL on `ex`. Refused unless `endpoint` is set, so it never creates a table in AWS; the production table belongs to your infrastructure code. | ### Process roles @@ -286,7 +309,17 @@ cache: version_ttl: 168h dedupe: - backend: pebble # in-process Pebble under /pebble + backend: pebble # in-process Pebble under /pebble; or dynamodb + lease: 30s # at most 59s with the embedded mq + reserve_concurrency: 64 + # dynamodb: # read only when backend is dynamodb + # table: wavehouse-dedupe-prod + # region: "" # empty = AWS_REGION + # endpoint: "" # dynamodb-local only + # timeout: 250ms + # max_attempts: 3 + # retry_mode: standard + # create_table: false # dynamodb-local only coord: backend: local # in-process leases (the sweeper's) @@ -362,6 +395,16 @@ WH_CACHE_REDIS_MAX_VALUE_BYTES=1048576 WH_CACHE_REDIS_COMPRESS_MIN_BYTES=1024 WH_CACHE_REDIS_VERSION_TTL=168h WH_DEDUPE_BACKEND=pebble +WH_DEDUPE_LEASE=30s +WH_DEDUPE_RESERVE_CONCURRENCY=64 +# Read only with WH_DEDUPE_BACKEND=dynamodb: +# WH_DEDUPE_DYNAMODB_TABLE=wavehouse-dedupe-prod +# WH_DEDUPE_DYNAMODB_REGION= +# WH_DEDUPE_DYNAMODB_ENDPOINT= +# WH_DEDUPE_DYNAMODB_TIMEOUT=250ms +# WH_DEDUPE_DYNAMODB_MAX_ATTEMPTS=3 +# WH_DEDUPE_DYNAMODB_RETRY_MODE=standard +# WH_DEDUPE_DYNAMODB_CREATE_TABLE=false WH_COORD_BACKEND=local WH_AUTH_JWT_SECRET=change-me-in-production diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 7f2e1e907..9c8029a44 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -175,7 +175,7 @@ WH_SETTINGS_DIR=/etc/wavehouse/settings WaveHouse keeps all embedded state under a single configurable root, `WH_DATA_DIR` (yaml: `data_dir`). Subdirectories are convention, not config: - `/nats` — embedded NATS JetStream. Holds in-flight events between an ingest POST and the ingest worker → ClickHouse flush, plus the `stream.gap_window_minutes` window (settings directory) of history that powers SSE gap-fill across restarts. -- `/pebble` — the Pebble dedup KV: one instance shared by every tenant, each key led by its tenant. Only used while some tenant's `dedupe.enabled` is `true` in its `config.json` (opened and closed on reload). +- `/pebble` — the Pebble dedup KV (with `dedupe.backend: pebble`, the default): one instance shared by every tenant, each key led by its tenant and table. Only used while some tenant's `dedupe.enabled` is `true` in its `config.json` (opened and closed on reload). It grows with every id kept: with `dedupe.retention` at `"0"` (forever) nothing is ever removed, so size the volume for it or set a [retention](/settings-directory#deduplication), whose expired ids an hourly sweep deletes. In a Docker / Podman / Kubernetes deployment, **`data_dir` must resolve to a host-backed volume**. The reference compose file `deployments/compose/standalone.yaml` sets `WH_DATA_DIR=/app/data` and binds a `wavehouse-data:/app/data` volume — copy that pattern. The bundled Dockerfiles pre-create `/app/data` and `/app/settings` owned by the nonroot user (UID 65532); the binary creates the `nats/` and `pebble/` subdirectories under `/app/data` itself on first run. @@ -183,7 +183,7 @@ If `data_dir` resolves into the container's writable overlay layer instead, **Je Beyond persistence, the *speed* of that volume matters: JetStream `fsync`s every event to `/nats` before the ingest endpoint returns `200`, so the volume's `fsync` latency is your ingest latency floor. Managed cloud block storage handles this without thinking; commodity or virtualized substrates (ZFS without a SLOG, qcow2-on-`ext4`, spinning disks) can stall ingest with multi-second `fsync` tails. See [Durability & Storage](/durability) to measure yours before going live. -WaveHouse runs a simple existence check on startup and logs a `WARN` if `/nats` (or `/pebble`, when dedupe is on) is missing or empty: +WaveHouse runs a simple existence check on startup and logs a `WARN` if `/nats` (or `/pebble`, when dedupe is on with the `pebble` backend) is missing or empty: ```text wrap=false WARN data directory does not exist — starting with no prior state. @@ -415,7 +415,7 @@ The folder name is the tenant id, and each folder is a complete settings directo **The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` takes it too, and reads a rejected or removed tenant's dead-letter queue like a served one's, since the queue is kept; a tenant that has none is a `404`. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). -**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Its message queue is its own as well: its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history that gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a rejected row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached results (`POST /v1/query` and pipe alike) dropped the moment it is back on one, so a repaired or restored folder never serves rows cached before the inserts it missed. A tenant whose folder moves it to another address or database has them dropped too, since they came from other tables. Apart from those two drops, a cached pipe result stays until its TTL expires, since a pipe names no table and no insert invalidates it. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. +**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant and table), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it, and in each table. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Its message queue is its own as well: its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history that gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a rejected row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached results (`POST /v1/query` and pipe alike) dropped the moment it is back on one, so a repaired or restored folder never serves rows cached before the inserts it missed. A tenant whose folder moves it to another address or database has them dropped too, since they came from other tables. Apart from those two drops, a cached pipe result stays until its TTL expires, since a pipe names no table and no insert invalidates it. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. **What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. Tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its seen ids kept for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its open `GET /v1/stream` connections are ended, as any tenant's are when it stops being served — the other tenants' events are untouched. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. @@ -425,9 +425,9 @@ The folder name is the tenant id, and each folder is a complete settings directo ## Multiple instances and the shared cache -Several WaveHouse instances can serve one ClickHouse behind a load balancer, but most of what each one holds is its own. The message queue is embedded, so an event is inserted by the instance that took its `POST /v1/ingest`, and reaches only that instance's SSE subscribers. The dedupe store is per instance too, so an id one instance has seen is new to another. +Several WaveHouse instances can serve one ClickHouse behind a load balancer, but most of what each one holds is its own. The message queue is embedded, so an event is inserted by the instance that took its `POST /v1/ingest`, and reaches only that instance's SSE subscribers. With the default `dedupe.backend: pebble` the dedupe store is per instance too, so an id one instance has seen is new to another; [`dedupe.backend: dynamodb`](#a-shared-dedupe-table-on-dynamodb) shares seen ids across instances. -The query-result cache is the layer that can be shared today. With the default `cache.backend: local`, each instance caches in its own memory, and an insert invalidates only the cache of the instance that made it. Every other instance keeps serving its cached results for the rows before the insert until each entry's TTL runs out, between 10 s and 1 h depending on how long the query took. With [`cache.backend: redis`](/configuration#cache), every instance reads and fills one Redis-compatible server, and an insert on any instance invalidates the cached results of every instance. The server is a standalone one or a Redis Cluster; Sentinel (`mode: sentinel`) refuses boot until [#656](https://github.com/Wave-RF/WaveHouse/issues/656), since the cache does not yet authenticate to the sentinels or refresh their topology. +The query-result cache and the dedupe store are the layers that can be shared today. With the default `cache.backend: local`, each instance caches in its own memory, and an insert invalidates only the cache of the instance that made it. Every other instance keeps serving its cached results for the rows before the insert until each entry's TTL runs out, between 10 s and 1 h depending on how long the query took. With [`cache.backend: redis`](/configuration#cache), every instance reads and fills one Redis-compatible server, and an insert on any instance invalidates the cached results of every instance. The server is a standalone one or a Redis Cluster; Sentinel (`mode: sentinel`) refuses boot until [#656](https://github.com/Wave-RF/WaveHouse/issues/656), since the cache does not yet authenticate to the sentinels or refresh their topology. **What another instance can see.** Ingest is already asynchronous: `/v1/ingest` answers before the batch is inserted. Once the inserting instance's worker has written the batch to ClickHouse, it replaces the table's version token in Redis, and from then on a lookup on any instance misses and reads the new rows. The cache adds no delay of its own beyond that single write. The exceptions: @@ -467,6 +467,93 @@ ORDER BY (page); WaveHouse discovers this schema on startup and refreshes it every `schema.refresh_interval` seconds (settings directory; seed default 60). You can also trigger an immediate refresh via `POST /v1/ops/schema/refresh` (admin-only). +## Upgrading across the dedupe key change + +The dedupe key now carries the table as well as the tenant ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)), so **an id deduped before the upgrade is not recognized after it**: a record carrying it is accepted once more. Nothing is migrated. The old keys never count as seen, and the dedupe sweep deletes them: its first pass runs about a minute after the instance opens, and `wavehouse_dedupe_swept_keys_total{reason="version_0"}` counts them ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)). Pebble returns their disk space as it compacts, not at once. Only a tenant with `dedupe.enabled` on is affected, and only by a record sent both before and after the upgrade — typically a producer retrying across the restart. To avoid duplicate rows, let retrying producers finish, or pause them, before upgrading. + +The same release adds an optional **`dedupe.retention`** key. No upgrade step is needed: a `config.json` without it keeps every id forever, as before. See [Deduplication](/settings-directory#deduplication) for a finite one. + +## A shared dedupe table on DynamoDB + +Pebble is per process, so two pods on it do not share seen ids. The DynamoDB backend keeps every tenant's ids in **one shared table**, and a conditional write makes a claim atomic across every pod that uses the table. WaveHouse **never creates this table in production**: the table belongs to your infrastructure code. The backend refuses to create a table unless it is pointed at a custom endpoint, so table creation only works against [dynamodb-local](https://docs.aws.amazon.com/amazondynamodb/latest/developerguide/DynamoDBLocal.html). + +What the backend requires of the table: + +| Attribute | Type | Role | +|---|---|---| +| `pk` | String | Partition key, and the only key: tenant, table and id as readable text, for example `acme/clicks/evt-123` (the table and id escaped the way NATS subject tokens are: letters, digits, `_` and `-` kept, every other byte written as `%XX`). No sort key. | +| `st` | Number | `1` = pending claim, `2` = committed. | +| `ex` | Number | Epoch seconds: the lease end while pending, the retention end once committed; absent = never expires. | +| `tk` | Binary | The claim token that `Release` matches. | + +Only `pk` is declared in the table definition. Turn TTL on for `ex`. Correctness never depends on TTL, because a claim whose `ex` has passed counts as absent whether or not DynamoDB has deleted it yet; TTL only reclaims the storage. TTL removes lapsed claims, and a committed id once its [`dedupe.retention`](/settings-directory#deduplication) ends. With the default retention `"0"` (forever) a committed item carries no `ex` and is kept, so the table grows by one item (about 200 bytes) per distinct id. Boot checks the table and logs a warning if TTL is off; a key schema that does not match is a misconfigured table, handled as described below. + +An example in Terraform. Replace the tags with your own conventions: + +```hcl +resource "aws_dynamodb_table" "wavehouse_dedupe" { + name = "wavehouse-dedupe-${var.environment}" + billing_mode = "PAY_PER_REQUEST" # provisioned + auto scaling once traffic is steady + hash_key = "pk" + deletion_protection_enabled = true + + attribute { + name = "pk" + type = "S" + } + + ttl { + attribute_name = "ex" + enabled = true + } + + server_side_encryption { + enabled = true + } + + tags = { + Name = "wavehouse-dedupe-${var.environment}" + Project = "wavehouse" + Environment = var.environment + ManagedBy = "terraform" + } +} + +# The pods' role (EKS Pod Identity or IRSA). No Scan, no CreateTable. +data "aws_iam_policy_document" "wavehouse_dedupe" { + statement { + actions = [ + "dynamodb:PutItem", + "dynamodb:DeleteItem", + "dynamodb:BatchWriteItem", + "dynamodb:DescribeTable", + "dynamodb:DescribeTimeToLive", + ] + resources = [aws_dynamodb_table.wavehouse_dedupe.arn] + } +} +``` + +Select it in the boot config, on every pod that should share seen ids (all the keys are in the [Configuration Reference](/configuration#dynamodb-dedupe)): + +```yaml +dedupe: + backend: dynamodb + dynamodb: + table: wavehouse-dedupe-prod + region: us-east-1 # or leave empty for AWS_REGION +``` + +or `WH_DEDUPE_BACKEND=dynamodb`, `WH_DEDUPE_DYNAMODB_TABLE=wavehouse-dedupe-prod`. A table that is missing, has the wrong key schema, or refuses the pod's credentials refuses boot over a flat settings directory whose tenant has dedupe on, and is logged at `ERROR` otherwise. In every other case — a throttle or network failure, a nested directory, or no tenant with dedupe on — the pod boots, every tenant with dedupe on (now or after a reload) fails its ingest closed, and the check is retried in the background (backing off from one second to thirty, and at once after every reload). A reload makes no table call itself, and does not wait on a tenant whose dedupe setting is unchanged; it waits only for a tenant whose store it closes — dedupe switched off, or the tenant removed or rejected — and then only for that tenant's in-flight calls, before its store closes. No region at all (neither `region` nor one from the SDK chain: `AWS_REGION`, `AWS_DEFAULT_REGION` or a profile) refuses boot in both shapes. The check runs in every pod running the `api` [role](/configuration#process-roles), whether or not any tenant has `dedupe.enabled` on; a pod without it opens no dedupe store. The per-tenant switch stays in each tenant's `config.json`. + +For development against dynamodb-local, set `dedupe.dynamodb.endpoint` (for example `http://localhost:8000`) and `create_table: true`, and give the SDK any static credentials (`AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`) and a region. `create_table` without an `endpoint` refuses boot. + +- **Credentials** come from the AWS SDK's default chain (EKS Pod Identity or IRSA in a pod; the environment or a profile locally), never from WaveHouse configuration. +- **Point-in-time recovery** is not needed. The table records which ids have been seen, so losing it produces duplicate rows, not lost events. +- **Cost:** every new event is two writes (the claim, then the commit), and a duplicate is one. On-demand, that is about $1.25 per million new events in us-east-1. Provisioned capacity with auto scaling is cheaper once traffic is steady. Storage is the other line: every distinct id stays in the table until its retention ends, forever at the default (see TTL above), at DynamoDB's per-GB-month rate. +- **One table serves every tenant,** so one tenant's burst can throttle the rest. A throttled or unreachable table fails the ingest request closed rather than publishing un-deduped. After five throttled or unreachable claims in a row within one second, the backend stops calling the table for a second and fails every tenant's dedupe requests immediately (`wavehouse_dedupe_dynamodb_short_circuits_total`). A duplicate or in-flight answer is not a failure and resets the count. +- **Metrics:** `wavehouse_dedupe_dynamodb_requests_total{op,outcome}`, `wavehouse_dedupe_dynamodb_request_duration_seconds{op}`, `wavehouse_dedupe_dynamodb_unprocessed_items_total`, `wavehouse_dedupe_dynamodb_short_circuits_total`. The table's own CloudWatch metrics `ThrottledRequests`, `SystemErrors` and `ConsumedWriteCapacityUnits` are worth alerting on too. + ## Upgrading across the v2 ingest envelope The NATS envelope changed shape in this release: the row now travels positionally, with `format`, `columns` and `row` replacing `data` — and the queue changed layout with it: boot deletes the earlier build's queue (below), so nothing an older version published reaches the new worker, which could not read it anyway (it carries no `format`, so there is no way to say which value belongs to which column). **Drain first** to keep what the old build had not yet inserted. diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index 22c8e6beb..a3d5b0d37 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -16,7 +16,7 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: | **Go** | 1.26+ (matches `go.mod`) | Compiles `cmd/wavehouse`; also runs the pinned `tool` deps (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `deadcode`, `gsa`, `goda`) via `go tool` | [go.dev/dl](https://go.dev/dl/) | | **GNU Make** | **4.0+** | The Makefile uses `--output-sync=target` (Make 4 only) and bash-pinned recipes. macOS ships with BSD Make 3.81, which **will not work** | macOS: `brew install make` then use `gmake` or put `$(brew --prefix make)/libexec/gnubin` on your PATH. Linux: usually already installed | | **bash** | 4+ recommended | Recipes are pinned to `bash`; the helper scripts under `scripts/` use `set -euo pipefail` and bash arrays | macOS default is bash 3.2 (works for current recipes, but `brew install bash` is safer); Linux distros ship 4+ | -| **Docker** *(or Podman)* | Engine 20.10+ with the Compose **v2** plugin (`docker compose`, no hyphen) | Compose stacks under `deployments/compose/`; the E2E and integration suites boot ClickHouse and a Redis via testcontainers (no compose file), and the integration suite also runs the shared cache backend against Redis, Valkey, Dragonfly (pulled from `docker.dragonflydb.io`) and a one-node Redis Cluster | [Docker Desktop](https://docs.docker.com/get-docker/), [colima](https://github.com/abiosoft/colima), or [Podman](https://podman.io) with `podman-compose` / the `podman compose` plugin. The testcontainers Go library also honors `DOCKER_HOST` for rootless Podman setups | +| **Docker** *(or Podman)* | Engine 20.10+ with the Compose **v2** plugin (`docker compose`, no hyphen) | Compose stacks under `deployments/compose/`; the E2E and integration suites boot ClickHouse and a Redis via testcontainers (no compose file), the integration suite also dynamodb-local, and the integration suite also runs the shared cache backend against Redis, Valkey, Dragonfly (pulled from `docker.dragonflydb.io`) and a one-node Redis Cluster | [Docker Desktop](https://docs.docker.com/get-docker/), [colima](https://github.com/abiosoft/colima), or [Podman](https://podman.io) with `podman-compose` / the `podman compose` plugin. The testcontainers Go library also honors `DOCKER_HOST` for rootless Podman setups | | **Node.js** | 22 LTS — pinned via `.nvmrc` at the repo root | Runtime for pnpm and the Vitest suites. Pinned to match CI (`setup-node` uses 22) and to avoid Node-major surprises; older Vitest versions in this repo were known to crash on Node 26 with a V8 heap-allocation abort | [nodejs.org](https://nodejs.org/) or `nvm use` / `fnm use` / `volta` (all read `.nvmrc`) | | **pnpm** | 11.21+ (pinned via `packageManager` in the root `package.json`) | Package manager for the TypeScript SDK, E2E test harness, and docs site (managed as a single pnpm workspace from the repo root); `make build-ts`, `make test-ts`, `make test-e2e`, `make build-docs`, `make dev-docs`, `make preview-docs` all shell out to `pnpm` | `corepack enable && corepack prepare pnpm@11.21.0 --activate` (recommended), or `npm i -g pnpm` | | **git** + **curl** | any recent | `git` for source + version metadata in builds; `curl` is used by the Makefile to fetch the pinned `golangci-lint` binary into `.bin/` | usually preinstalled | @@ -345,9 +345,9 @@ Each test target writes `covdata` to `tmp/coverage//data/`, renders a tex | E2E tests (SDK) | `tests/e2e/sdk/*.test.ts` | Yes | `make test-e2e` | - **Unit tests** live beside the code they test (e.g., `internal/discovery/discovery_test.go`). They use mocks or embedded NATS (in-process, no Docker needed). -- **Integration tests** use the `//go:build integration` build tag. In `tests/integration`, `TestMain` starts one ClickHouse testcontainer and boots the production wiring against it through `app.New` (embedded NATS, ingest worker, sweeper, hub, the API server on a random loopback port); tests reach it via `env(t)` and create their own tables. DLQ tests use `assert.Eventually` with a 30-second timeout for the 5-second ingest worker batch window. `internal/cache`'s integration tests start their own containers instead — Redis, Valkey, Dragonfly and a one-node Redis Cluster — for the shared backend. `shared_cache_test.go` starts its own Redis testcontainer per test (`startRedis`) and boots extra, independent `cache.backend: redis` instances over that same ClickHouse (`bootRedisApp`), to exercise the cache shared across processes rather than one package in isolation. +- **Integration tests** use the `//go:build integration` build tag. In `tests/integration`, `TestMain` starts one ClickHouse testcontainer and a dynamodb-local one (for the DynamoDB dedupe backend's tests), and boots the production wiring against it through `app.New` (embedded NATS, ingest worker, sweeper, hub, the API server on a random loopback port); tests reach it via `env(t)` and create their own tables. DLQ tests use `assert.Eventually` with a 30-second timeout for the 5-second ingest worker batch window. `internal/cache`'s integration tests start their own containers instead — Redis, Valkey, Dragonfly and a one-node Redis Cluster — for the shared backend. `shared_cache_test.go` starts its own Redis testcontainer per test (`startRedis`) and boots extra, independent `cache.backend: redis` instances over that same ClickHouse (`bootRedisApp`), to exercise the cache shared across processes rather than one package in isolation. -Shared test utilities live in `internal/testutil/`. The packages log through `slog.Default()`, so tests reach log output through `internal/testutil/logtest`: `logtest.Silence()` in a package's `TestMain` discards it, and `logtest.Capture(t, level)` routes it to a buffer for a test that asserts on log lines — such a test must not call `t.Parallel()`, because the default logger is process-wide. +Shared test utilities live in `internal/testutil/`. The packages log through `slog.Default()`, so tests reach log output through `internal/testutil/logtest`: `logtest.Silence()` in a package's `TestMain` discards it, and `logtest.Capture(t, level)` routes it to a buffer for a test that asserts on log lines — such a test must not call `t.Parallel()`, because the default logger is process-wide. A test that starts the embedded broker keeps its store in `internal/testutil/storedir`'s `storedir.New(t)` rather than a bare `t.TempDir()` (`testutil.NewEmbeddedMQ` does): the NATS server can finish writing a consumer's state after `Close` returns, which fails `t.TempDir`'s one-shot removal, and `storedir` removes the store again until those writes have landed ([#442](https://github.com/Wave-RF/WaveHouse/issues/442)). ### Adding New Tests @@ -460,10 +460,10 @@ WaveHouse/ │ ├── chsql/ # Shared ClickHouse SQL helpers (quoting + bind-safety) │ ├── config/ # YAML + env var configuration │ ├── coord/ # Leases with fencing tokens (in-process Local, RunElected, coordtest suite) -│ ├── dedupe/ # Optional deduplication (Pebble) +│ ├── dedupe/ # Optional deduplication (Reserve/Commit/Release; Pebble or DynamoDB) │ ├── discovery/ # ClickHouse schema introspection + validation │ ├── ingest/ # Batch buffering + DLQ + Active Sweeper -│ ├── keyenc/ # One escaping for composite keys (NATS subject tokens, cache keys) +│ ├── keyenc/ # One escaping for composite keys (NATS subject tokens, cache keys, dedupe keys) │ ├── mq/ # MQ boundary: the only NATS/JetStream importer │ ├── observability/ # OpenTelemetry pipeline (traces/metrics/logs + Prometheus) │ ├── pipes/ # Named query pipes (types + parameter binding) diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index 8e57d8231..a67b3a756 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -58,6 +58,14 @@ The strict guarantee translates well to managed cloud infrastructure — the pre The tell for a commit-cadence problem (ZFS-without-SLOG, noisy-neighbor VM host) is that a single-threaded benchmark looks fine while a concurrent one is far worse — so always benchmark with multiple writers, and benchmark the guest **and** the host if virtualized. +## Deduplication: one more fsync per window + +With [deduplication](/settings-directory#deduplication) on, a `200` also means the records' ids were committed to the dedupe store, or, if that commit failed, that the failure was counted by `wavehouse_ingest_dedupe_commit_failed_total` and the ids lapse with their lease. On the embedded Pebble store that commit is an `fsync` of its own. It is taken once per window of up to 256 records of a request, after the window's publishes, rather than once per record: a 1,000-record batch costs four dedupe syncs, not a thousand. Measured with `BenchmarkIngest_DedupBatchOnPebble` on a developer laptop, with the queue stubbed out so only the dedupe store touched disk, the dedupe work for that batch took 24 ms windowed against 5.7 s one record at a time; the JetStream publishes' own fsyncs come on top. A single-record request still pays one sync for its publish and one for its commit. + +With a finite `dedupe.retention` and `dedupe.backend: pebble`, expired ids are deleted by a background sweep, an hour apart; on DynamoDB the table's TTL deletes them instead ([Deployment](/deployment#a-shared-dedupe-table-on-dynamodb)). Its deletes are not fsynced (a delete lost to a crash is redone by the next pass), so it adds no sync to the ingest path. It reads 1,024 keys at a time without holding up commits, then re-reads the expired ones and deletes those still expired; a commit waits only for that last step, at most 1,024 point reads and one unsynced write, however many deleted keys the read stepped over. An expired id is already treated as new by the next claim of it, sweep or no sweep, so retention never depends on the sweep having run. + +A publish can also fail after JetStream stored the event (a timeout on the ack). The record's id is then left to lapse with its dedupe lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default) rather than given back, and every deduped record is published under an idempotency key derived from its tenant, table and id, which each tenant's ingest stream remembers for two minutes after the first publish. A retry after the lease but inside those two minutes is therefore dropped by the stream rather than stored twice; one later than that is stored again. The duplicate window has to cover more than the lease alone: the `503` for an uncertain publish sends the *full* lease as `Retry-After`, so an obedient client's retry can land up to ~2×lease after the original request, and a claim's expiry can itself round up by a further second on some backends — the invariant is `lease + ceil(lease) + 1s ≤ window` (`ceil` rounding up to the whole second, so `2×lease + 1s` for a whole-second lease), not just `lease ≤ window`, and boot refuses a `dedupe.lease` that breaks it with the embedded queue, so at most 59 seconds. Two minutes against the default 30-second lease clears that with room to spare. For the same reason a finite `dedupe.retention` must be at least those two minutes: an id re-sent after a shorter retention ended would be claimed again, then dropped by the stream as a copy while the client was told it was accepted. Settings validation refuses one below it. + ## Check your storage before you trust it Replicate JetStream's exact pattern — a 4 KiB write followed by a flush, in a tight loop — and report the percentiles. The numbers that matter are **p99** and **max**: those are your worst-case publish latency. diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index 321c37291..9d0a9ba75 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -40,7 +40,7 @@ The SDK **never throws** for anything the server returns — all API errors come | 502 | `clickhouse.misconfigured` | No | ClickHouse refused WaveHouse's own credentials or database, or the route to it is wrong (a redirect, or a `4xx` other than `408`/`413`/`429`, with no exception code) — an operator fix | | 502 | `clickhouse.response_too_large` | No | A raw-SQL (`wh.sql`) response over the 64 MiB cap | | 503 | `clickhouse.unavailable` | Yes | ClickHouse is down, unreachable or overloaded; `Retry-After: 5`, honored between attempts | -| 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, a schema not discovered yet, a tenant on no ClickHouse pool, or a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on that last cause waits the 30 s; a stream re-dials on its own jittered backoff instead | +| 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, a schema not discovered yet, a tenant on no ClickHouse pool, a dedupe store that cannot answer (`dedupe store unavailable`, `Retry-After: 5`), a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`), or a record whose dedupe id another request is still publishing (`a request with the same dedupe id is in flight`, `Retry-After`: the server's dedupe lease, 30 s by default). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on those last two causes waits that long; a stream re-dials on its own jittered backoff instead | | 0 | `NETWORK_ERROR` | Yes | Network failure (retried with exponential backoff) | | 0 | `ABORTED` | No | Request canceled via `AbortSignal` | | 0 | `SSE_CONNECT_ERROR` | No | Stream could not be started (e.g. a non-absolute `baseURL`) | diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index ced176e67..0a390de46 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -11,7 +11,7 @@ Boot config — the YAML file and `WH_*` environment variables on the [Configura The settings directory holds WaveHouse's file-based settings as exactly four JSON documents: [`roles.json`](#rolesjson), [`policies.json`](#policiesjson), [`pipes.json`](#pipesjson), and [`config.json`](#configjson-keys). The files are the only write path — standalone, you edit them on the host; on WaveHouse Cloud the control plane writes them — and there is no API that writes back to them. Every file must exist (an empty document is `{}` — a missing file always means deletion or a wrong path, never "defaults"), and any other entry in the directory is an error, so a typoed filename or a stray backup fails loudly instead of being silently ignored. Dot-prefixed entries are the one exception: editor swap files and the `..data` machinery Kubernetes ConfigMap mounts publish through are ignored. -Create one with `wavehouse bootstrap [dir]`: it writes all four files with every key at its default and refuses a non-empty directory, so an existing settings directory is never overwritten. The binary carries no compiled defaults — the seed is the one place they live, and what the server adopts is exactly what the files say. The seed ships no policy (`policies.json` is `{}`, `roles.json` and `pipes.json` are empty lists), so a freshly bootstrapped directory boots fail-closed — every request is denied until you write a policy. The container images ship no settings directory: they preset `WH_SETTINGS_DIR=/app/settings` and expect a bind mount there — a host directory you wrote with `bootstrap` (the reference compose file mounts the checked-in `deployments/compose/settings/`, the seed with `clickhouse.addr` pointed at the `clickhouse` service and a permissive `public` trial policy in `policies.json` / `roles.json`). A bind mount, not a named volume: the images are distroless, with no shell to edit files inside a volume. A missing mount refuses to boot rather than running on defaults nobody chose. +Create one with `wavehouse bootstrap [dir]`: it writes all four files with every key at its default and refuses a non-empty directory, so an existing settings directory is never overwritten. The binary carries no compiled defaults but one — a missing `dedupe.retention` means `"0"`, forever — so the seed is where the defaults live, and what the server adopts is exactly what the files say. The seed ships no policy (`policies.json` is `{}`, `roles.json` and `pipes.json` are empty lists), so a freshly bootstrapped directory boots fail-closed — every request is denied until you write a policy. The container images ship no settings directory: they preset `WH_SETTINGS_DIR=/app/settings` and expect a bind mount there — a host directory you wrote with `bootstrap` (the reference compose file mounts the checked-in `deployments/compose/settings/`, the seed with `clickhouse.addr` pointed at the `clickhouse` service and a permissive `public` trial policy in `policies.json` / `roles.json`). A bind mount, not a named volume: the images are distroless, with no shell to edit files inside a volume. A missing mount refuses to boot rather than running on defaults nobody chose. Check a directory with `wavehouse validate [dir]`. Both commands resolve the directory the same way — the argument, falling back to `WH_SETTINGS_DIR`, and a usage error (exit `2`) with neither — so the path you seed is the path you validate, and inside the container images (which preset `WH_SETTINGS_DIR=/app/settings`) both work with no argument at all. `validate` validates without starting the server (JSON syntax including unknown fields and duplicate keys, per-file shape rules including the required keys, and cross-file role references), prints every finding in one pass, and exits `0` for valid (warnings allowed), `1` for invalid, `2` for usage — so operators and CI can gate a settings change before it reaches a running instance. @@ -99,7 +99,7 @@ Pipes are read per request, so a reload changes what the next `GET /v1/pipes/{na ## `config.json` keys -The tenant tunables. Every key is required (a missing one is a validation error) except the per-table overrides; the "Seed" column is what `wavehouse bootstrap` writes: +The tenant tunables. Every key is required (a missing one is a validation error) except `dedupe.retention` and the per-table overrides; the "Seed" column is what `wavehouse bootstrap` writes: | Key | Seed | Description | | --- | ---- | ----------- | @@ -123,7 +123,8 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `dedupe.enabled` | `false` | Turn deduplication on; a reload opens or closes this tenant's store — see [Deduplication](#deduplication). | | `dedupe.id_field` | `event_id` | Dedup key field — see [Deduplication](#deduplication). | | `dedupe.require_id` | `false` | Reject rows missing the id field — see [Deduplication](#deduplication). | -| `dedupe.tables.
.{id_field, require_id}` | `{}` | Optional per-table overrides; each entry overrides only the fields it names and inherits the rest. | +| `dedupe.retention` | `"0"` | How long a committed id stays a duplicate, as a duration (`"720h"`); `"0"`, or leaving the key out, keeps it forever — see [Deduplication](#deduplication). | +| `dedupe.tables.
.{id_field, require_id, retention}` | `{}` | Optional per-table overrides; each entry overrides only the fields it names and inherits the rest. | | `dlq.enabled` | `true` | Park poison rows — those ClickHouse still rejects after row-by-row isolation, and every row of a batch whose tenant has no ClickHouse connection — on the tenant's dead-letter stream (`DLQ_{tenant}`) (`false`: leave them unacked for redelivery — except an envelope the worker cannot read, which is dropped and counted) — see [Dead Letter Queue](#dead-letter-queue). | | `dlq.tables.
.enabled` | `{}` | Optional per-table override of the switch. | | `query.timestamp_bucket_seconds` | `60` | Bucket (seconds, `>= 0`) that a structured query's relative time range is truncated to, so near-identical queries share a cache entry; `0` disables bucketing. Read per query. | @@ -161,8 +162,9 @@ The tenant tunables. Every key is required (a missing one is a validation error) "enabled": false, "id_field": "event_id", "require_id": false, + "retention": "720h", "tables": { - "clicks": { "id_field": "click_id" } + "clicks": { "id_field": "click_id", "retention": "24h" } } }, "dlq": { @@ -179,16 +181,17 @@ The tenant tunables. Every key is required (a missing one is a validation error) } ``` -What stays in boot config is only what cannot change under a running process — the implementation each layer runs on (`mq.backend`, `cache.backend`, `dedupe.backend`, `coord.backend`) and a shared backend's connection (`cache.redis`), the process's `roles`, resource sizing (`data_dir`, `cache.l1_max_cost`, `clickhouse.max_total_conns`), the listeners, the observability exporters — and the **secrets**: `clickhouse.password`, `cache.redis.password`, `auth.jwt_secret`, `auth.operator_key`. Secrets never belong in a tracked JSON file, so they stay in the environment and are combined with the wiring here on every (re)connect; rotating one is a restart. See [Configuration](/configuration). Everything else lives here and reloads. +What stays in boot config is only what cannot change under a running process — the implementation each layer runs on (`mq.backend`, `cache.backend`, `dedupe.backend`, `coord.backend`) and a shared backend's connection (`cache.redis`, `dedupe.dynamodb`), how a dedupe claim behaves (`dedupe.lease`, `dedupe.reserve_concurrency`), the process's `roles`, resource sizing (`data_dir`, `cache.l1_max_cost`, `clickhouse.max_total_conns`), the listeners, the observability exporters — and the **secrets**: `clickhouse.password`, `cache.redis.password`, `auth.jwt_secret`, `auth.operator_key`. Secrets never belong in a tracked JSON file, so they stay in the environment and are combined with the wiring here on every (re)connect; rotating one is a restart. See [Configuration](/configuration). Everything else lives here and reloads. ## Deduplication -Every dedupe knob lives here — there are no boot-config keys for it. The switch and its fields are resolved per record from one snapshot (table override → global value): +Every per-tenant dedupe knob lives here. Where the seen ids are kept (`dedupe.backend`) and how long a claim is held (`dedupe.lease`) are [boot config](/configuration#dedupe), the same for every tenant. The switch and its fields are resolved per record from one snapshot (table override → global value): -- `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens or closes this tenant's store in the embedded Pebble instance at `/pebble`, so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`500 dedupe failed`) until the next reload or restart — the files asked for dedupe, so publishing un-deduped is not a fallback. At boot a failed open refuses to start, like every other store. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) every tenant's seen ids live in that one instance, each key led by its tenant, and it is open while any tenant's switch is on: each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and keeps its seen ids for the folder that restores it; and if that instance fails to open, at boot or on reload, every tenant with dedupe on fails closed — its ingest answers `500 dedupe failed` until a reload opens it — while the tenants with dedupe off carry on. -- `dedupe.id_field` (seed default `event_id`) — JSON field name in the ingest body used as the dedup key. -- `dedupe.require_id` (seed default `false`) — controls what happens to a row missing `id_field` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. -- `dedupe.tables.
.{id_field, require_id}` — per-table overrides; each entry overrides only the fields it names and inherits the rest. +- `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens or closes this tenant's store (in the embedded Pebble instance at `/pebble`, or its share of the DynamoDB table under `dedupe.backend: dynamodb`), so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`503 dedupe store unavailable`, `Retry-After: 5`) until it opens — the files asked for dedupe, so publishing un-deduped is not a fallback. With `dedupe.backend: pebble` that is the next reload or restart, and at boot a failed open over a flat directory refuses to start, like every other store; with `dynamodb` it is the background retry described below. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) with `dedupe.backend: pebble`, every tenant's seen ids live in that one instance, each key led by its tenant and table, and it is open while any tenant's switch is on: each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and keeps its seen ids for the folder that restores it; and if that instance fails to open, at boot or on reload, every tenant with dedupe on fails closed — its ingest answers `503 dedupe store unavailable` (`Retry-After: 5`) until a reload opens it — while the tenants with dedupe off carry on. Under `dedupe.backend: dynamodb` the table check plays the instance's part, in either shape: the table is checked whether or not any tenant's switch is on, and a table that fails it fails every tenant with dedupe on closed until the check, retried in the background and at once after every reload, passes. Only a misconfigured table (missing, the wrong key schema, access denied) over a flat directory whose tenant has dedupe on refuses boot instead ([Configuration](/configuration#dynamodb-dedupe)). +- `dedupe.id_field` (seed default `event_id`) — JSON field name in the ingest body used as the dedup key. An id is a duplicate only within its own tenant and table: the same value in two tables is two ids. An id longer than 1,024 bytes once escaped (every byte but an ASCII letter, digit, `_` or `-` takes three) is stored as its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. While its record is being published, an id is held for its lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default): another request carrying the same id meanwhile gets `503` (`a request with the same dedupe id is in flight`) with the lease, in whole seconds, as `Retry-After` — see [the ingest errors](/api#post-v1ingesttabletable--ingest-data). An id is committed only after its record is published; if that commit fails (counted by `wavehouse_ingest_dedupe_commit_failed_total`, which should stay at zero), the record is still answered `ok` and the id lapses with its lease: a retry of it before then answers in-flight, one inside the ingest queue's two-minute duplicate window is dropped there by its idempotency key, and one after that is stored again. +- `dedupe.require_id` (seed default `false`) — controls what happens to a row missing `id_field`, or carrying it as `null` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. +- `dedupe.retention` (optional; seed default `"0"`) — how long a committed id stays a duplicate, as a Go duration string: `"24h"`, `"720h"` (30 days), `"90m"`. There is no day unit. `"0"` keeps every id forever, which was the only behavior before this key existed, and a `config.json` without the key means the same. Once an id's retention has ended, the next record carrying it is published as new. With `dedupe.backend: pebble`, a background sweep over the shared Pebble instance deletes the expired id: first about a minute after the instance opens (when the first tenant switches dedupe on), then hourly while any tenant keeps it on, counted by `wavehouse_dedupe_swept_keys_total{reason="expired"}`. With `dynamodb`, no sweep runs: the table's TTL on `ex` deletes the item ([Deployment](/deployment#a-shared-dedupe-table-on-dynamodb)). A finite retention must be at least `"2m"`, the ingest queue's duplicate window: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and then dropped by the queue as a copy, while the client was told it was accepted. A retention below that is refused, not raised to the minimum; so are a negative value and anything that is not a duration, such as `"30d"`, a number with no unit (`"300"` needs one: `"300s"`; `"0"` is the one exception), or a JSON number rather than a string. Hot-reloadable: a change applies to ids committed after the reload, and an id already committed keeps the expiry it was stored with. +- `dedupe.tables.
.{id_field, require_id, retention}` — per-table overrides; each entry overrides only the fields it names and inherits the rest, so a table with no `retention` keeps the tenant's (forever when the tenant sets none). A table can keep ids for a shorter time than its tenant, or for longer, or forever (`"retention": "0"`) under a finite tenant retention. ## ClickHouse diff --git a/go.mod b/go.mod index d75355341..ee399126b 100644 --- a/go.mod +++ b/go.mod @@ -18,6 +18,11 @@ require ( github.com/ClickHouse/clickhouse-go/v2 v2.48.0 github.com/MicahParks/jwkset v0.11.3 github.com/MicahParks/keyfunc/v3 v3.8.2 + github.com/aws/aws-sdk-go-v2 v1.47.1 + github.com/aws/aws-sdk-go-v2/config v1.33.6 + github.com/aws/aws-sdk-go-v2/credentials v1.20.6 + github.com/aws/aws-sdk-go-v2/service/dynamodb v1.69.1 + github.com/aws/smithy-go v1.28.1 github.com/cockroachdb/pebble v1.1.5 github.com/dgraph-io/ristretto/v2 v2.4.2 github.com/dustin/go-humanize v1.0.1 @@ -76,6 +81,17 @@ require ( github.com/andybalholm/brotli v1.2.2 // indirect github.com/antithesishq/antithesis-sdk-go v0.7.2-default-no-op // indirect github.com/aws/aws-sdk-go v1.49.4 // indirect + github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1 // indirect + github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4 // indirect + github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4 // indirect + github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/endpoint-discovery v1.13.4 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4 // indirect + github.com/aws/aws-sdk-go-v2/service/signin v1.10.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sso v1.38.1 // indirect + github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sts v1.51.1 // indirect github.com/aymanbagabas/go-osc52/v2 v2.0.1 // indirect github.com/beorn7/perks v1.0.1 // indirect github.com/bitfield/gotestdox v0.2.2 // indirect diff --git a/go.sum b/go.sum index 3b203e221..bf6e503ba 100644 --- a/go.sum +++ b/go.sum @@ -45,6 +45,38 @@ github.com/antithesishq/antithesis-sdk-go v0.7.2-default-no-op h1:p2zFsAzvhIpFya github.com/antithesishq/antithesis-sdk-go v0.7.2-default-no-op/go.mod h1:FQyySiasQQM8735Ddel3MRojmy4dA1IqCeyJ5jmPMbI= github.com/aws/aws-sdk-go v1.49.4 h1:qiXsqEeLLhdLgUIyfr5ot+N/dGPWALmtM1SetRmbUlY= github.com/aws/aws-sdk-go v1.49.4/go.mod h1:LF8svs817+Nz+DmiMQKTO3ubZ/6IaTpq3TjupRn3Eqk= +github.com/aws/aws-sdk-go-v2 v1.47.1 h1:uOIZnp4PK3ZhKI0dNrJrhTEsLxbpXHTAJlwoS1pvAtw= +github.com/aws/aws-sdk-go-v2 v1.47.1/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU= +github.com/aws/aws-sdk-go-v2/config v1.33.6 h1:MBjkSTLczek/UgiK+EYPIoRTqE7gP8vtW3OFbFo7Nug= +github.com/aws/aws-sdk-go-v2/config v1.33.6/go.mod h1:grRAFzdAZJrwcbasJRg2MPvIrVjtlfXllHssN6+E1JE= +github.com/aws/aws-sdk-go-v2/credentials v1.20.6 h1:NpAFXCU7NzXNkdGK3zQTtsRJ+3v9tZQV0xcdRw8uBdw= +github.com/aws/aws-sdk-go-v2/credentials v1.20.6/go.mod h1:mcZCoiPnyMvP8VMNbygNX5lLqSlkYJIMPODylQMurOk= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1 h1:8gALAAmacnIXh+z6VkdDanv4/IkG5APdg4DZLDTmLog= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1/go.mod h1:Z7IJhJU+poOdJjUR2wpyY21ossQ1XS/R3Lk9Msq5kM4= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4 h1:CLq4+8UHCI+ZZYl/EuJxXovaIVN2xeeT8JV+dsApQ5E= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4/go.mod h1:Wv4q5sAM04xAMkoOedxLx2inVf6K5FdxYp+A61L+q/0= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4 h1:dD4MR81I7YkpEBRk6UP9rocC2QnT3qVuXwzlYTtfGEs= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4/go.mod h1:EcXV1kAFd5XwSkDHlj94gnF3q5CkJyYiIJfH8N0VmrE= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4 h1:7Wo47d/xn/7KttCSBd8EGYeZ7ULRFRkUHr6vkZPBzVQ= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4/go.mod h1:tDB2IVC1xC3vX8o+6uRlzhTxP3g1b77CZXFX/oD2FnQ= +github.com/aws/aws-sdk-go-v2/service/dynamodb v1.69.1 h1:bKwiQA6SKqFXBO+1IwP/hTwCU5RlqeitG4gVvSuMN8U= +github.com/aws/aws-sdk-go-v2/service/dynamodb v1.69.1/go.mod h1:Gm+i2GlUsFNlzoBq8VXF44XHbKANn3tV8nYBBp3rN8Q= +github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 h1:bAdDl/HkGCcGPoe25ToSHEw23VIxt6CT5fLcg111BKg= +github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19/go.mod h1:KaUzbLxv4CeSxh6ZCl9B4m7CuFenS8kUEaDs+f/DQr4= +github.com/aws/aws-sdk-go-v2/service/internal/endpoint-discovery v1.13.4 h1:6HvmOQ1rBRrZ4qPJSWxd5szPKUsngXCwSw+V3UaJHmw= +github.com/aws/aws-sdk-go-v2/service/internal/endpoint-discovery v1.13.4/go.mod h1:zv2N29aiQUhG2XZNM9zgwCnAyVBdTBbcIpfNAlNmA20= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4 h1:29SvnfGhXjTl8ONxFwbj2rs6lbhiFXD2CgFQmbT/bXY= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4/go.mod h1:wm04I5DMuNVvZHFe/dHnUxincvNbbK7AiNBbYsQivek= +github.com/aws/aws-sdk-go-v2/service/signin v1.10.1 h1:DzCCWLzcIRQ77F3DEUljud7bEjTgFOIKXP52NmVRyhU= +github.com/aws/aws-sdk-go-v2/service/signin v1.10.1/go.mod h1:xpo/geVldu8payT375WekctUzopG/hBU7miiqItMUlw= +github.com/aws/aws-sdk-go-v2/service/sso v1.38.1 h1:Umtl/0YZhng4xndfW3lKJrYYP7NLEjI6bGXVomwLcs0= +github.com/aws/aws-sdk-go-v2/service/sso v1.38.1/go.mod h1:rRD/dnm7q0HYE/I5TMaPgkWyyUGLcwuxHLABsLnQ3e0= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1 h1:orIWdNiLgzrhu/11RcPPKO/SBzUUymbUQuZbSPImghg= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1/go.mod h1:skwM/xsbR/1ReUTesv9BhpJp1VjajR7DWQnuVLwiXsQ= +github.com/aws/aws-sdk-go-v2/service/sts v1.51.1 h1:0HOqZXRvMytH6bFHVIc0oJX07sZjfhz0zXtjs6gdE8s= +github.com/aws/aws-sdk-go-v2/service/sts v1.51.1/go.mod h1:26zA0GhDrLo+yiLI2yXWxqB1PdsShfLikoI7GOEgugM= +github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ= +github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= github.com/aymanbagabas/go-osc52/v2 v2.0.1 h1:HwpRHbFMcZLEVr42D4p7XBqjyuxQH5SMiErDT4WkJ2k= github.com/aymanbagabas/go-osc52/v2 v2.0.1/go.mod h1:uYgXzlJ7ZpABp8OJ+exZzJJhRNQ2ASbcXHWsFqH8hp8= github.com/aymanbagabas/go-udiff v0.3.1 h1:LV+qyBQ2pqe0u42ZsUEtPiCaUoqgA9gYRDs3vj1nolY= diff --git a/internal/api/ingest.go b/internal/api/ingest.go index ca89bb83e..fa1ea816e 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -7,8 +7,10 @@ import ( "fmt" "io" "log/slog" + "math" "net/http" "sort" + "strconv" "strings" "time" @@ -35,6 +37,11 @@ import ( // with the admin query handler — see internal/api/query.go. const maxReportedResults = 10000 +// ingestWindow is how many records a batch prepares before reserving, +// publishing and committing them together: one dedupe call per phase per +// window rather than per record, and at most one window of encoded rows held. +const ingestWindow = 256 + // IngestHandler handles POST /v1/ingest?table={table} type IngestHandler struct { // Registry yields the request tenant's schema registry. @@ -43,14 +50,16 @@ type IngestHandler struct { // store, picked off the store the handler already holds (#583 story 7; // dedupe.Stores in production). nil when no dedupe store is wired (tests). Dedup func(store *settings.Store) dedupe.Deduplicator - // DedupeSettings resolves the effective dedupe id_field/require_id for a - // table of the request's tenant ((*settings.Store).DedupeFor in - // production). Called once per record so a settings reload lands at a - // record boundary — one record never mixes two documents' values. Dedup is - // skipped when nil. - DedupeSettings func(store *settings.Store, table string) (enabled bool, idField string, requireID bool) - Publisher mq.Publisher - PolicySource PolicySource + // DedupeSettings resolves the effective dedupe settings for a table of the + // request's tenant ((*settings.Store).DedupeFor in production). Called + // once per record so a settings reload lands at a record boundary — one + // record never mixes two documents' values. Dedup is skipped when nil. + DedupeSettings func(store *settings.Store, table string) settings.Dedupe + // DedupeLease is how long a record's claimed id stays pending while it is + // published; 0 means dedupe.DefaultLease. + DedupeLease time.Duration + Publisher mq.Publisher + PolicySource PolicySource // Validator and Checker are the per-record seams a native type layer will // take over (see ingest_seams.go). Both are optional: nil means the default @@ -63,6 +72,8 @@ type IngestHandler struct { // tests can pin the cap-overflow path without allocating 16 MiB per run; not // a production tuning knob, hence unexported. Mirrors QueryHandler. maxRequestBytes int64 + // window overrides ingestWindow when > 0, for tests and benchmarks. + window int } func NewIngestHandler(registry RegistrySource, pub mq.Publisher) *IngestHandler { @@ -74,6 +85,13 @@ var dedupeMissingIDCounter, _ = otel.Meter("wavehouse-ingest").Int64Counter( metric.WithDescription("Ingested records missing the configured dedupe id_field (idempotency skipped)"), ) +// dedupeCommitFailedCounter counts records published whose id could not be +// committed afterwards: a retry after the lease lapses publishes them again. +var dedupeCommitFailedCounter, _ = otel.Meter("wavehouse-ingest").Int64Counter( + "wavehouse_ingest_dedupe_commit_failed_total", + metric.WithDescription("Published records whose dedupe id failed to commit afterwards (the claim lapses with its lease)"), +) + // dedupeDisabledCounter counts records published un-deduped because the // settings snapshot said dedupe was on while the store was switched off — // transient across a reload; a climbing rate means the store and the @@ -121,12 +139,15 @@ type recordReject struct { // requestAbort is a whole-request failure: this record and every one that // follows is refused. Both paths stop and return the status; the batch path -// abandons the remaining records rather than silently losing the tail. +// abandons the remaining records rather than silently losing the tail. What +// earlier windows published stays published, and with dedupe on stays +// committed, so a whole-batch retry reports those records as duplicates. // // Most causes are TRANSIENT system conditions, where abandoning the tail is what // makes the batch safe to retry: publish backpressure (503), an unreachable -// broker (503, mq.ErrUnavailable), a publish/marshal failure (500), a dedup -// backend error (500). +// broker (503, mq.ErrUnavailable), a publish/marshal failure (500), a dedupe +// store that cannot answer (503) or fails (500), an id another request holds +// (503). // // One is not. An insert grant that resolved for the other operation is a 403 and // a caller/config bug — retrying cannot help. It aborts rather than rejecting @@ -136,7 +157,7 @@ type recordReject struct { type requestAbort struct { Status int Message string - RetryAfter string // non-empty → emit a Retry-After header (503: backpressure or an unavailable broker) + RetryAfter string // non-empty → emit a Retry-After header } func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { @@ -311,16 +332,21 @@ func (h *IngestHandler) handleSingle( return } - dup, reject, abort := h.processRecord(ctx, store, table, scope, schema, perms, role, data, now, checkGuard) + rec, abort := h.prepareRecord(ctx, store, table, scope, schema, perms, role, data, now, checkGuard) + if abort == nil && rec.reject == nil { + window := []pendingRecord{rec} + abort = h.ingestWindow(ctx, store, table, scope, window) + rec = window[0] + } if abort != nil { writeAbort(w, abort) return } - if reject != nil { - writeJSONError(w, reject.Status, reject.Message) + if rec.reject != nil { + writeJSONError(w, rec.reject.Status, rec.reject.Message) return } - if dup { + if rec.duplicate { w.Header().Set("Content-Type", "application/json") _ = json.NewEncoder(w).Encode(map[string]bool{"duplicate": true}) return @@ -352,6 +378,24 @@ func (h *IngestHandler) handleBatch( checkGuard *recordReject, ) { result := batchResult{Results: []recordResult{}} + size := h.window + if size <= 0 { + size = ingestWindow + } + window := make([]pendingRecord, 0, min(size, 16)) + // flush runs the window's records through reserve → publish → commit and + // reports them in order; false when it aborted the request. + flush := func() bool { + if abort := h.ingestWindow(ctx, store, table, scope, window); abort != nil { + writeAbort(w, abort) + return false + } + for i := range window { + result.add(&window[i]) + } + window = window[:0] + return true + } for { data, err := rr.Next() @@ -361,8 +405,10 @@ func (h *IngestHandler) handleBatch( if err != nil { if rse, ok := errors.AsType[*recordSyntaxError](err); ok { result.Total++ - result.Failed++ - appendResult(&result, recordResult{Index: result.Total, Error: rse.Error()}) + window = append(window, pendingRecord{index: result.Total, reject: &recordReject{Message: rse.Error()}}) + if len(window) == size && !flush() { + return + } continue } // Unreachable while the body is buffered — a bytes.Reader cannot produce @@ -381,26 +427,22 @@ func (h *IngestHandler) handleBatch( } result.Total++ - idx := result.Total - dup, reject, abort := h.processRecord(ctx, store, table, scope, schema, perms, role, data, now, checkGuard) + rec, abort := h.prepareRecord(ctx, store, table, scope, schema, perms, role, data, now, checkGuard) if abort != nil { // Whole-request failure: surface the status rather than recording a // request-scoped condition as per-record loss (see requestAbort). + // Nothing in the open window has been published. writeAbort(w, abort) return } - if reject != nil { - result.Failed++ - appendResult(&result, recordResult{Index: idx, Error: reject.Message}) - continue - } - if dup { - result.Duplicates++ - appendResult(&result, recordResult{Index: idx, Duplicate: true}) - continue + rec.index = result.Total + window = append(window, rec) + if len(window) == size && !flush() { + return } - result.Succeeded++ - appendResult(&result, recordResult{Index: idx, Ok: true}) + } + if len(window) > 0 && !flush() { + return } slog.InfoContext(ctx, "batch ingested", "table", table, @@ -410,12 +452,24 @@ func (h *IngestHandler) handleBatch( _ = json.NewEncoder(w).Encode(result) } -// appendResult records a per-record outcome up to maxReportedResults. The -// batchResult counts are incremented by the caller and stay authoritative even -// when the Results slice is truncated. -func appendResult(result *batchResult, entry recordResult) { - if len(result.Results) < maxReportedResults { - result.Results = append(result.Results, entry) +// add counts rec's outcome and records it up to maxReportedResults; the counts +// stay authoritative when Results is truncated. Total is counted as records +// are read. +func (r *batchResult) add(rec *pendingRecord) { + entry := recordResult{Index: rec.index} + switch { + case rec.reject != nil: + r.Failed++ + entry.Error = rec.reject.Message + case rec.duplicate: + r.Duplicates++ + entry.Duplicate = true + default: + r.Succeeded++ + entry.Ok = true + } + if len(r.Results) < maxReportedResults { + r.Results = append(r.Results, entry) } } @@ -451,7 +505,7 @@ func writeMaxBytesError(w http.ResponseWriter, err error, limit int64) bool { // // Evaluated here rather than per record because the condition is a property of // (table, role, policy) and is identical for every record in the request — the -// same reasoning as the !resolved abort in processRecord. Doing it per record +// same reasoning as the !resolved abort in prepareRecord. Doing it per record // would emit one ERROR line per record for a single mis-wired policy, which on // a 16 MiB body of small records is ~1.2M lines. The reject is still returned // per record, so a batch reports each record's own cause: one that SUPPLIES the @@ -464,7 +518,7 @@ func (h *IngestHandler) policyCheckGuard( ) *recordReject { checks, resolved := perms.CheckClauses() if !resolved { - return nil // the !resolved abort in processRecord owns this case + return nil // the !resolved abort in prepareRecord owns this case } // Sorted, and every offender — not the first one a map range happens to @@ -512,19 +566,32 @@ func (h *IngestHandler) policyCheckGuard( } } -// processRecord runs the per-record pipeline shared by the single-object and -// batch ingest paths: schema validation → column/check permission enforcement -// (with claim-derived auto-injection) → optional dedup → publish. The -// table-level insert grant is checked once by the caller before any record is -// processed, so perms here drives only the per-column and per-row checks (it is -// nil when no policy store is configured). data may be mutated to auto-inject -// check-clause values. +// pendingRecord is one record between prepareRecord and its outcome. +type pendingRecord struct { + index int // 1-based position in a batch + reject *recordReject // non-nil: the record is bad and is not published + payload []byte // the encoded envelope to publish + // key is the record's dedupe identity, nil when it is published + // un-deduped; retention is how long its id stays a duplicate once + // committed; claim is Reserve's answer for it. + key *dedupe.Key + retention time.Duration + claim dedupe.Claim + duplicate bool +} + +// prepareRecord runs the per-record half of the pipeline shared by the +// single-object and batch ingest paths: schema validation → column/check +// permission enforcement (with claim-derived auto-injection) → timestamp +// canonicalization → dedupe id resolution → encoding. Reserving, publishing +// and committing happen per window, in ingestWindow. The table-level insert +// grant is checked once by the caller before any record is processed, so perms +// here drives only the per-column and per-row checks (it is nil when no policy +// store is configured). data may be mutated to auto-inject check-clause values. // -// Exactly one of the outcomes is meaningful per call: -// - duplicate true: the record was skipped by dedup (reject/abort nil). -// - reject non-nil: the record is bad; the rest of a batch may still proceed. -// - abort non-nil: a whole-request failure; the caller stops and returns it. -func (h *IngestHandler) processRecord( +// A record the rest of a batch may proceed past comes back with reject set; +// abort non-nil is a whole-request failure the caller stops and returns. +func (h *IngestHandler) prepareRecord( ctx context.Context, store *settings.Store, table, scope string, @@ -534,10 +601,10 @@ func (h *IngestHandler) processRecord( data map[string]any, now time.Time, checkGuard *recordReject, -) (duplicate bool, reject *recordReject, abort *requestAbort) { +) (rec pendingRecord, abort *requestAbort) { if err := h.validator().Validate(schema, data); err != nil { slog.WarnContext(ctx, "schema validation failed", "error", err, "table", table) - return false, &recordReject{Status: http.StatusBadRequest, Message: err.Error()}, nil + return pendingRecord{reject: &recordReject{Status: http.StatusBadRequest, Message: err.Error()}}, nil } // DEEP AUTH: column-level allow/deny + check clauses. @@ -545,10 +612,10 @@ func (h *IngestHandler) processRecord( for col := range data { if !perms.IsColumnAllowed(col, true) { slog.WarnContext(ctx, "column insertion forbidden", "column", col, "role", role) - return false, &recordReject{ + return pendingRecord{reject: &recordReject{ Status: http.StatusForbidden, Message: fmt.Sprintf("column %q not allowed for insert", col), - }, nil + }}, nil } } // Through the accessor, not a bare read. The check loop iterates a side's @@ -569,7 +636,7 @@ func (h *IngestHandler) processRecord( // permission failures for one mis-wired grant. slog.ErrorContext(ctx, "insert checks consulted on a grant resolved for another operation", "table", table, "role", role) - return false, nil, &requestAbort{ + return pendingRecord{}, &requestAbort{ Status: http.StatusForbidden, Message: "insert permissions were not resolved for this request", } @@ -584,7 +651,7 @@ func (h *IngestHandler) processRecord( // a record that supplies the column fails schema validation first with // a different message, and a batch should report each its own cause. if checkGuard != nil { - return false, checkGuard, nil + return pendingRecord{reject: checkGuard}, nil } // A []any value is an _in check: the inserted value must be present and // one of the allowed set. Unlike the scalar _eq case there is no single @@ -593,10 +660,10 @@ func (h *IngestHandler) processRecord( actual, ok := data[col] if !ok || !h.checker().InSet(actual, set) { slog.WarnContext(ctx, "check clause failed", "column", col, "allowed", set, "actual", actual, "present", ok) - return false, &recordReject{ + return pendingRecord{reject: &recordReject{ Status: http.StatusForbidden, Message: fmt.Sprintf("check failed for column %q", col), - }, nil + }}, nil } continue } @@ -613,10 +680,10 @@ func (h *IngestHandler) processRecord( // reading the token's own JSON type didn't give it. if !h.checker().Matches(actual, requiredVal) { slog.WarnContext(ctx, "check clause failed", "column", col, "expected", requiredVal, "actual", actual) - return false, &recordReject{ + return pendingRecord{reject: &recordReject{ Status: http.StatusForbidden, Message: fmt.Sprintf("check failed for column %q", col), - }, nil + }}, nil } } else { // Auto-inject the required value if not provided — as a plain @@ -636,45 +703,31 @@ func (h *IngestHandler) processRecord( // enforces) after the permission checks: check clauses keep pre-#372 semantics. h.validator().CanonicalizeTimestamps(schema, data) - // Optional deduplication. enabled/id_field/require_id resolve per record + // Optional deduplication. The dedupe settings resolve per record // from one snapshot (table override → global; the settings directory - // always states them, so no compiled fallback is needed), so a reload - // lands at a record boundary. A Deduplicator without a settings source is - // a wiring bug, not a mode — main wires both or neither. + // states them all but dedupe.retention, whose absence means "0"), so a + // reload lands at a record boundary. A Deduplicator without a settings source is + // a wiring bug, not a mode — main wires both or neither. The id is claimed + // in ingestWindow, once every record of the window is encoded, so nothing + // but the publish can fail while the claim is held. if h.Dedup != nil && h.DedupeSettings != nil { - if enabled, idField, requireID := h.DedupeSettings(store, table); enabled { - idVal, ok := data[idField] - if !ok { + if dd := h.DedupeSettings(store, table); dd.Enabled { + idField := dd.IDField + // An explicit null is as missing as an absent key (#370): fmt.Sprint + // would make every null "", one id for every such record. + if idVal, ok := data[idField]; ok && idVal != nil { + rec.key = &dedupe.Key{Table: table, ID: fmt.Sprint(idVal)} + rec.retention = dd.Retention + } else { dedupeMissingIDCounter.Add(ctx, 1, metric.WithAttributes(attribute.String("table", table))) - if requireID { - slog.WarnContext(ctx, "dedupe id_field missing; rejecting", "id_field", idField, "table", table) - return false, &recordReject{ + if dd.RequireID { + slog.WarnContext(ctx, "dedupe id_field missing or null; rejecting", "id_field", idField, "table", table) + return pendingRecord{reject: &recordReject{ Status: http.StatusBadRequest, Message: fmt.Sprintf("missing dedupe id field %q", idField), - }, nil - } - slog.WarnContext(ctx, "dedupe id_field missing; publishing without idempotency", "id_field", idField, "table", table) - } else { - eventID := fmt.Sprint(idVal) - dup, err := h.Dedup(store).CheckAndMark(ctx, eventID) - switch { - case errors.Is(err, dedupe.ErrDisabled): - // A reload flipped dedupe.enabled between the snapshot - // read above and this call (the two transition at - // different instants). Publish un-deduped, as a record - // under the other setting would have been. The counter - // carries the signal (a burst is a reload; a steady rate - // is the store and settings out of step), so the line is - // Debug rather than a WARN per record. - dedupeDisabledCounter.Add(ctx, 1, metric.WithAttributes(attribute.String("table", table))) - slog.DebugContext(ctx, "dedupe switched off mid-reload; publishing without idempotency", "event_id", eventID, "table", table) - case err != nil: - slog.ErrorContext(ctx, "dedupe check failed", "error", err, "event_id", eventID) - return false, nil, &requestAbort{Status: http.StatusInternalServerError, Message: "dedupe failed"} - case dup: - slog.InfoContext(ctx, "duplicate event skipped", "event_id", eventID) - return true, nil, nil + }}, nil } + slog.WarnContext(ctx, "dedupe id_field missing or null; publishing without idempotency", "id_field", idField, "table", table) } } } @@ -687,7 +740,7 @@ func (h *IngestHandler) processRecord( row, err := ingest.EncodeCompactRow(cols, data) if err != nil { slog.ErrorContext(ctx, "failed to encode compact row", "error", err, "table", table) - return false, nil, &requestAbort{Status: http.StatusInternalServerError, Message: "marshal failed"} + return pendingRecord{}, &requestAbort{Status: http.StatusInternalServerError, Message: "marshal failed"} } evt := ingest.EventMessage{ @@ -699,28 +752,221 @@ func (h *IngestHandler) processRecord( Row: row, } - payload, err := json.Marshal(evt) + rec.payload, err = json.Marshal(evt) if err != nil { slog.ErrorContext(ctx, "failed to marshal event message", "error", err) - return false, nil, &requestAbort{Status: http.StatusInternalServerError, Message: "marshal failed"} + return pendingRecord{}, &requestAbort{Status: http.StatusInternalServerError, Message: "marshal failed"} + } + return rec, nil +} + +// ingestWindow reserves, publishes and commits one window of prepared +// records, in three phases: one Reserve for every keyed record, the publishes +// in record order, one Commit for every claim published. Rejected and +// duplicate records are skipped. It sets each record's outcome and returns an +// abort when the request must stop; what the window published before a failure +// is committed first, so the retry reports it as duplicates (see publishFailed). +func (h *IngestHandler) ingestWindow(ctx context.Context, store *settings.Store, table, scope string, recs []pendingRecord) *requestAbort { + var dd dedupe.Deduplicator + var keyed []int + for i := range recs { + if recs[i].reject == nil && recs[i].key != nil { + keyed = append(keyed, i) + } + } + if len(keyed) > 0 { + dd = h.Dedup(store) + if abort := h.reserve(ctx, dd, table, recs, keyed); abort != nil { + return abort + } + } + + topic := mq.Topic{Tenant: store.Tenant(), Table: table, Scope: scope} + for i := range recs { + rec := &recs[i] + if rec.reject != nil || rec.duplicate { + continue + } + var opts []mq.PublishOpt + if rec.claim.Status == dedupe.Claimed { + // The retry of an uncertain publish carries the same id, so the + // queue drops its copy if the first one landed. + opts = append(opts, mq.WithIdempotencyKey(dedupe.IdempotencyKey(store.Tenant(), rec.claim.Key))) + } + if err := h.Publisher.Publish(ctx, topic, rec.payload, opts...); err != nil { + return h.publishFailed(ctx, dd, topic, recs, i, err) + } + } + commitClaims(ctx, dd, recs, table) + return nil +} + +// lease is the dedupe lease in effect: h.DedupeLease when set, else +// dedupe.DefaultLease. Shared by reserve (Reserve's argument) and +// publishFailed (the Retry-After of a claim left to lapse), so both name the +// same window a client is told to wait out. +func (h *IngestHandler) lease() time.Duration { + if h.DedupeLease > 0 { + return h.DedupeLease + } + return dedupe.DefaultLease +} + +// reserve claims the keys of recs[keyed] in one call and records each answer. +// A duplicate is skipped. A key another request holds releases the window's +// claims and aborts with 503 and the lease as Retry-After, since that +// request's outcome decides this one's. A store that cannot answer now is a +// 503 too; nothing in the window has been published. ErrDisabled — a reload +// switched the store off after the settings snapshot was read — publishes the +// window un-deduped, as records under the other setting would have been. +func (h *IngestHandler) reserve(ctx context.Context, dd dedupe.Deduplicator, table string, recs []pendingRecord, keyed []int) *requestAbort { + lease := h.lease() + keys := make([]dedupe.Key, len(keyed)) + for j, i := range keyed { + keys[j] = *recs[i].key + } + claims, err := dd.Reserve(ctx, keys, lease) + switch { + case errors.Is(err, dedupe.ErrDisabled): + // The counter carries the signal (a burst is a reload; a steady rate + // is the store and settings out of step), so the line is Debug rather + // than a WARN per record. + dedupeDisabledCounter.Add(ctx, int64(len(keys)), metric.WithAttributes(attribute.String("table", table))) + slog.DebugContext(ctx, "dedupe switched off mid-reload; publishing without idempotency", "records", len(keys), "table", table) + return nil + case errors.Is(err, dedupe.ErrUnavailable): + slog.WarnContext(ctx, "dedupe store unavailable", "error", err, "table", table) + return &requestAbort{Status: http.StatusServiceUnavailable, Message: "dedupe store unavailable", RetryAfter: "5"} + case err != nil: + if ctx.Err() != nil { + // The request's own context ended — the client is gone, or its + // deadline passed — while Reserve was in flight. Reserve wraps + // that as an ordinary error, but it is not a backend problem + // worth an operator's attention, and the response status below + // is moot: nothing is listening for it. + slog.DebugContext(ctx, "dedupe reserve failed: request context ended", "error", err, "table", table) + } else { + slog.ErrorContext(ctx, "dedupe reserve failed", "error", err, "table", table) + } + return &requestAbort{Status: http.StatusInternalServerError, Message: "dedupe failed"} + } + var held *dedupe.Key + for j, i := range keyed { + recs[i].claim = claims[j] + switch claims[j].Status { + case dedupe.Duplicate: + recs[i].duplicate = true + slog.InfoContext(ctx, "duplicate event skipped", "event_id", keys[j].ID, "table", table) + case dedupe.InFlight: + if held == nil { + held = &keys[j] + } + case dedupe.Claimed: + } + } + if held != nil { + releaseClaims(ctx, dd, claimedIn(recs)) + slog.InfoContext(ctx, "event id in flight in another request", "event_id", held.ID, "table", table) + return &requestAbort{ + Status: http.StatusServiceUnavailable, + Message: "a request with the same dedupe id is in flight", + RetryAfter: strconv.Itoa(int(math.Ceil(lease.Seconds()))), + } + } + return nil +} + +// publishFailed settles a window whose publish failed at recs[k] and returns +// the abort. The records before k are queued, so their ids are committed. A +// definite failure — ErrQueueFull, the broker refused the event — releases k's +// id and the rest, so the client's retry publishes them (#384). Any other +// failure may have stored the event before failing, so k's claim is left to +// lapse with its lease instead: a retry before then answers in-flight, and one +// after republishes under the same idempotency key, which the queue drops if +// the first copy landed. The records after k were never sent and are +// released. +// +// mq.ErrUnavailable — a broker blip, not a refusal — is one such uncertain +// failure, but still answers 503 rather than the plain 500 below: when k held +// a Claimed claim (left to lapse, as above), Retry-After is that lease +// rounded up to whole seconds, so an obedient client waits out the in-flight +// window instead of retrying straight into it and getting the 503 reserve +// already answers for that; when k was never keyed there is no lapse to wait +// out, so Retry-After is the flat 5 seconds main's per-record path used. +func (h *IngestHandler) publishFailed(ctx context.Context, dd dedupe.Deduplicator, topic mq.Topic, recs []pendingRecord, k int, err error) *requestAbort { + definite := errors.Is(err, mq.ErrQueueFull) + commitClaims(ctx, dd, recs[:k], topic.Table) + after := k + 1 + if definite { + after = k + } + releaseClaims(ctx, dd, claimedIn(recs[after:])) + switch { + case definite: + slog.WarnContext(ctx, "ingest queue is full", "tenant", topic.Tenant, "error", err, "table", topic.Table, "scope", topic.Scope) + return &requestAbort{Status: http.StatusServiceUnavailable, Message: "service unavailable", RetryAfter: "30"} + case errors.Is(err, mq.ErrUnavailable): + retryAfter := "5" + if recs[k].claim.Status == dedupe.Claimed { + retryAfter = strconv.Itoa(int(math.Ceil(h.lease().Seconds()))) + } + slog.WarnContext(ctx, "ingest queue unavailable", "tenant", topic.Tenant, "error", err, "table", topic.Table, "scope", topic.Scope) + return &requestAbort{Status: http.StatusServiceUnavailable, Message: "service unavailable", RetryAfter: retryAfter} + } + slog.ErrorContext(ctx, "failed to publish to the ingest queue", "tenant", topic.Tenant, "error", err, "table", topic.Table, "scope", topic.Scope) + return &requestAbort{Status: http.StatusInternalServerError, Message: "publish failed"} +} + +// claimedIn is the Claimed claims among recs. +func claimedIn(recs []pendingRecord) []dedupe.Claim { + var out []dedupe.Claim + for i := range recs { + if recs[i].claim.Status == dedupe.Claimed { + out = append(out, recs[i].claim) + } } + return out +} - slog.DebugContext(ctx, "publishing event to the ingest queue", "table", table, "scope", scope) - if err := h.Publisher.Publish(ctx, mq.Topic{Tenant: store.Tenant(), Table: table, Scope: scope}, payload); err != nil { - if errors.Is(err, mq.ErrQueueFull) { - slog.WarnContext(ctx, "ingest queue is full", "tenant", store.Tenant(), "error", err, "table", table, "scope", scope) - return false, nil, &requestAbort{Status: http.StatusServiceUnavailable, Message: "service unavailable", RetryAfter: "30"} +// commitClaims makes the ids of recs' Claimed claims duplicates, one Commit +// per retention — one in practice, unless a reload changed it mid-window. A +// failure does not fail the records — they are in the queue — so it is logged +// and counted, and the claims lapse after their lease. +func commitClaims(ctx context.Context, dd dedupe.Deduplicator, recs []pendingRecord, table string) { + var retentions []time.Duration + byRetention := map[time.Duration][]dedupe.Claim{} + for i := range recs { + if recs[i].claim.Status != dedupe.Claimed { + continue + } + r := recs[i].retention + if _, ok := byRetention[r]; !ok { + retentions = append(retentions, r) } - if errors.Is(err, mq.ErrUnavailable) { - // A broker blip, not a full queue: a sooner retry is likely to land. - slog.WarnContext(ctx, "ingest queue unavailable", "tenant", store.Tenant(), "error", err, "table", table, "scope", scope) - return false, nil, &requestAbort{Status: http.StatusServiceUnavailable, Message: "service unavailable", RetryAfter: "5"} + byRetention[r] = append(byRetention[r], recs[i].claim) + } + for _, r := range retentions { + claims := byRetention[r] + // The records are queued whatever the request's context does next. + err := dd.Commit(context.WithoutCancel(ctx), claims, r) + switch { + case err == nil, errors.Is(err, dedupe.ErrDisabled): + default: + dedupeCommitFailedCounter.Add(ctx, int64(len(claims)), metric.WithAttributes(attribute.String("table", table))) + slog.ErrorContext(ctx, "dedupe commit failed after publish; the ids lapse with their lease", "error", err, "table", table, "records", len(claims)) } - slog.ErrorContext(ctx, "failed to publish to the ingest queue", "tenant", store.Tenant(), "error", err, "table", table, "scope", scope) - return false, nil, &requestAbort{Status: http.StatusInternalServerError, Message: "publish failed"} } +} - return false, nil, nil +// releaseClaims gives back claims whose records were not published. A failure +// is only logged: the claims lapse with their lease either way. +func releaseClaims(ctx context.Context, dd dedupe.Deduplicator, claims []dedupe.Claim) { + if len(claims) == 0 { + return + } + if err := dd.Release(context.WithoutCancel(ctx), claims); err != nil && !errors.Is(err, dedupe.ErrDisabled) { + slog.WarnContext(ctx, "dedupe release failed; the ids lapse with their lease", "error", err) + } } // checkValueMatches decides insert-check equality: the payload value must diff --git a/internal/api/ingest_retention_test.go b/internal/api/ingest_retention_test.go new file mode 100644 index 000000000..ef65e4b2f --- /dev/null +++ b/internal/api/ingest_retention_test.go @@ -0,0 +1,104 @@ +package api + +import ( + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/mq" + "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil" +) + +// A finite retention must outlast the queue's duplicate window, or an id +// re-sent after it expires is claimed again and then dropped by the queue as +// a copy of the first publish. +func TestIngest_MinDedupeRetentionCoversTheDuplicateWindow(t *testing.T) { + t.Parallel() + assert.GreaterOrEqual(t, settings.MinDedupeRetention, mq.EmbeddedDuplicateWindow) +} + +// dedupeConfig is fullConfig with dedupe switched on and the given dedupe +// block's retention settings. +func dedupeConfig(retention, tables string) string { + return strings.Replace(fullConfig(100), + `"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}`, + `"dedupe": {"enabled": true, "id_field": "event_id", "require_id": false, "retention": "`+retention+`", "tables": `+tables+`}`, 1) +} + +// Each record is committed with its table's retention from the adopted +// settings, and a reload changes it for the next request: the retention is +// read per record, like id_field, not fixed when the store was opened. +func TestIngest_Dedup_CommitsWithTheAdoptedRetention(t *testing.T) { + t.Parallel() + dir := writeSettingsFixture(t, dedupeConfig("720h", `{"users": {"retention": "0"}}`)) + tenants, findings := settings.Open(dir) + require.NotNil(t, tenants, "findings: %v", findings) + store, _ := tenants.For(tenant.Default) + + reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ + {Name: "clicks", Columns: []discovery.Column{{Name: "event_id", Type: "String"}}}, + {Name: "users", Columns: []discovery.Column{{Name: "event_id", Type: "String"}}}, + }) + dedup := testutil.NewMockDeduplicator() + h := NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}) + h.Dedup = staticDedup(dedup) + h.DedupeSettings = (*settings.Store).DedupeFor + ingest := func(table, id string) { + t.Helper() + w := httptest.NewRecorder() + req := ingestRequest(t, table, map[string]any{"event_id": id}) + h.Handle(w, req.WithContext(WithStore(req.Context(), store))) + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + } + + ingest("clicks", "e1") + ingest("users", "e1") + assert.Equal(t, 720*time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "e1"})) + assert.Equal(t, time.Duration(0), dedup.Retention(dedupe.Key{Table: "users", ID: "e1"}), "the table keeps ids forever") + + require.NoError(t, os.WriteFile(filepath.Join(dir, settings.FileConfig), []byte(dedupeConfig("24h", `{}`)), 0o600)) + _, adopted := tenants.Reload("test") + require.True(t, adopted) + ingest("clicks", "e2") + ingest("users", "e2") + assert.Equal(t, 24*time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "e2"})) + assert.Equal(t, 24*time.Hour, dedup.Retention(dedupe.Key{Table: "users", ID: "e2"}), "the override is gone") + assert.Equal(t, 720*time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "e1"}), "ids committed before the change keep theirs") +} + +// A reload that lands mid-window splits the window's commit by retention, so +// every record keeps the retention of the snapshot it was prepared under. +func TestIngest_Dedup_ReloadMidWindowCommitsEachRetention(t *testing.T) { + t.Parallel() + dedup := testutil.NewMockDeduplicator() + h := NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}) + h.Dedup = staticDedup(dedup) + var calls atomic.Int32 + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + if calls.Add(1) <= 2 { + return settings.Dedupe{Enabled: true, IDField: "event_id", Retention: time.Hour} + } + return settings.Dedupe{Enabled: true, IDField: "event_id", Retention: 2 * time.Hour} + } + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", + `{"page": "/", "event_id": "a"}`, `{"page": "/", "event_id": "b"}`, `{"page": "/", "event_id": "c"}`))) + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + assert.Equal(t, 2, dedup.Commits, "one Commit per retention") + assert.Equal(t, time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "a"})) + assert.Equal(t, time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "b"})) + assert.Equal(t, 2*time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "c"})) +} diff --git a/internal/api/ingest_seams.go b/internal/api/ingest_seams.go index d95fdb58f..210f2ba59 100644 --- a/internal/api/ingest_seams.go +++ b/internal/api/ingest_seams.go @@ -20,7 +20,7 @@ import ( // return would invite a caller to change that. // // The two are one interface because they are one contract — "what this schema -// says about this record" — evaluated at two points in processRecord that must +// says about this record" — evaluated at two points in prepareRecord that must // stay apart: the insert-check block sits between them deliberately, so checks // keep pre-#372 semantics. type RecordValidator interface { @@ -55,7 +55,7 @@ func (h *IngestHandler) validator() RecordValidator { // InsertChecker decides whether a record's value satisfies a policy check // clause. Matches answers the scalar `_eq` form (the required value), InSet the // `_in` form (set membership). It never sees a record as a whole: the -// auto-injection of a missing check value stays in processRecord, where the +// auto-injection of a missing check value stays in prepareRecord, where the // ordering against validation and canonicalization is load-bearing. type InsertChecker interface { Matches(actual, required any) bool diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index a87a4df92..233e36138 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -11,6 +11,7 @@ import ( "net/http/httptest" "net/url" "strings" + "sync" "testing" "testing/iotest" "time" @@ -200,7 +201,9 @@ func TestIngest_Dedup_FirstTime(t *testing.T) { dedup := testutil.NewMockDeduplicator() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "evt-1"}) w := httptest.NewRecorder() @@ -216,7 +219,9 @@ func TestIngest_Dedup_Duplicate(t *testing.T) { dedup := testutil.NewMockDeduplicator() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } // First call. req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "dup-1"}) @@ -673,7 +678,7 @@ func TestIngest_Policy_CheckIn_Absent_FailsClosed(t *testing.T) { // TestIngest_Policy_CheckIn_AbsentClaim_FailsClosed locks the typed-nil []any // path behind an _in check: when the claim itself is absent, resolveInValues -// returns a typed-nil []any, which must still assert as []any in processRecord +// returns a typed-nil []any, which must still assert as []any in prepareRecord // (entering the membership branch) so the column is rejected — never treated as a // scalar _eq value and auto-injected. The sibling _Absent test omits the column // with the claim present; this one drops the claim too. Guards #224 fail-closed. @@ -716,7 +721,9 @@ func TestIngest_DedupIsTheTenants(t *testing.T) { pub := &testutil.MockPublisher{} h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = func(s *settings.Store) dedupe.Deduplicator { return stores.For(s.Tenant()) } - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } ingest := func(id tenant.ID) string { store, ok := tenants.For(id) @@ -739,7 +746,9 @@ func TestIngest_Dedup_MissingIDField(t *testing.T) { dedup := testutil.NewMockDeduplicator() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } // Payload omits event_id and require_id is off: the row skips // dedup and is still published — the warn+counter path, not a rejection (#219). @@ -758,7 +767,9 @@ func TestIngest_Dedup_RequireID_Rejects(t *testing.T) { pub := &testutil.MockPublisher{} h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(testutil.NewMockDeduplicator()) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} + } w := httptest.NewRecorder() h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home"}))) @@ -780,7 +791,9 @@ func TestIngest_NDJSON_RequireID_Rejects(t *testing.T) { pub := &testutil.MockPublisher{} h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(testutil.NewMockDeduplicator()) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} + } req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a", "event_id": "e1"}), @@ -1028,7 +1041,9 @@ func TestIngest_NDJSON_Dedup(t *testing.T) { dedup := testutil.NewMockDeduplicator() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a", "event_id": "e1"}), @@ -1852,8 +1867,8 @@ func TestIngest_JSONArray_SyntaxError_Fatal(t *testing.T) { h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // A structural syntax error desyncs the decoder — the whole request fails - // (400), unlike a per-element type error. The leading good element may have - // already published (at-least-once on retry). + // (400), unlike a per-element type error. The leading good element is still + // in the open window, which is dropped unpublished. req := rawIngestRequest(t, "clicks", "application/json", `[{"page":"/a"}, {bad]`) w := httptest.NewRecorder() h.Handle(w, withTenant(req)) @@ -1861,7 +1876,7 @@ func TestIngest_JSONArray_SyntaxError_Fatal(t *testing.T) { assert.Equal(t, http.StatusBadRequest, w.Code) assert.Contains(t, w.Body.String(), "invalid json") testutil.AssertJSONErrorResponse(t, w) - assert.Len(t, pub.Messages, 1) // the leading record published before the abort + assert.Empty(t, pub.Messages) } func TestIngest_JSONArray_Truncated_Fatal(t *testing.T) { @@ -2333,7 +2348,9 @@ func TestIngest_Dedup_DisabledBySettings(t *testing.T) { dedup.Err = errors.New("must not be called while disabled") h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return false, "event_id", true } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{IDField: "event_id", RequireID: true} + } w := httptest.NewRecorder() h.Handle(w, withTenant(ingestRequest(t, "clicks", tt.body))) @@ -2353,7 +2370,9 @@ func TestIngest_Dedup_DisabledMidReload(t *testing.T) { dedup.Err = dedupe.ErrDisabled h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} + } w := httptest.NewRecorder() h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"event_id": "e1", "page": "/home"}))) @@ -2374,7 +2393,7 @@ func TestIngest_Dedup_DisabledMidReload(t *testing.T) { // discovery.Validate accepts `{}` here because every column is nullable or // defaulted. I previously asserted this path was unreachable, having tested only // against a schema with a required column; it is not. -func TestProcessRecord_UnresolvedInsertSideAborts(t *testing.T) { +func TestPrepareRecord_UnresolvedInsertSideAborts(t *testing.T) { t.Parallel() schema := &discovery.TableSchema{ Name: "loose", @@ -2398,11 +2417,10 @@ func TestProcessRecord_UnresolvedInsertSideAborts(t *testing.T) { require.NoError(t, discovery.Validate(schema, map[string]any{}), "all-nullable/defaulted columns accept an empty record — this is what makes the read reachable") - dup, reject, abort := h.processRecord( + rec, abort := h.prepareRecord( context.Background(), testStore, "loose", "", schema, selectResolved, "viewer", map[string]any{}, time.Now(), nil) - assert.False(t, dup) - assert.Nil(t, reject, "a request-scoped condition must not be reported per record") + assert.Nil(t, rec.reject, "a request-scoped condition must not be reported per record") require.NotNil(t, abort, "an unresolved insert side must abort the request") assert.Equal(t, http.StatusForbidden, abort.Status) assert.Empty(t, abort.RetryAfter, "not a transient condition — retrying cannot help") @@ -2765,3 +2783,351 @@ func TestIngest_CheckOnEphemeralColumn_Rejected(t *testing.T) { assert.Contains(t, jsonErrorMessage(t, w), "is ephemeral and is never stored") assert.Empty(t, pub.Messages, "an unenforceable check must publish nothing") } + +// dedupHandler is a handler over the clicks registry with dedupe on for +// event_id and dedup as the store. +func dedupHandler(t *testing.T, pub *testutil.MockPublisher, dedup dedupe.Deduplicator, requireID bool) *IngestHandler { + t.Helper() + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h.Dedup = staticDedup(dedup) + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: requireID} + } + return h +} + +// #384: a publish the queue refused gives the id back, so the retry the 503 +// asks for is published rather than skipped as a duplicate of a record that +// never reached the queue. +func TestIngest_Dedup_FailedPublishReleasesTheID(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull)} + dedup := testutil.NewMockDeduplicator() + h := dedupHandler(t, pub, dedup, false) + body := map[string]any{"page": "/home", "event_id": "e1"} + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", body))) + require.Equal(t, http.StatusServiceUnavailable, w.Code) + assert.Equal(t, "30", w.Header().Get("Retry-After")) + assert.False(t, dedup.Pending(dedupe.Key{Table: "clicks", ID: "e1"}), "released, not left to lapse") + + pub.Err = nil + w = httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", body))) + require.Equal(t, http.StatusOK, w.Code) + assert.Contains(t, w.Body.String(), `"ok":true`, "the retry is published, not a duplicate") + assert.Len(t, pub.Published(), 1) + + w = httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", body))) + assert.Contains(t, w.Body.String(), `"duplicate":true`, "and committed once published") +} + +// A publish whose outcome is unknown may have stored the event, so its claim +// is neither released nor committed: it lapses with the lease, a retry before +// then answers in-flight, and the idempotency key covers one after. +// mq.ErrUnavailable is such a failure too, and answers 503 immediately with +// the lease itself as Retry-After (rounded up to whole seconds) rather than +// the flat 5 seconds a request with no claim to lapse would get — the same +// distinction TestIngest_Dedup_FailedPublishReleasesTheID draws for the one +// failure (mq.ErrQueueFull) that releases instead. +func TestIngest_Dedup_UncertainPublishLeavesTheClaim(t *testing.T) { + t.Parallel() + tests := []struct { + name string + err error + status int + retryAfter string + }{ + {"unrecognized error", context.DeadlineExceeded, http.StatusInternalServerError, ""}, + {"unavailable broker", fmt.Errorf("%w: timeout", mq.ErrUnavailable), http.StatusServiceUnavailable, "30"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{Err: tt.err} + dedup := testutil.NewMockDeduplicator() + h := dedupHandler(t, pub, dedup, false) + body := map[string]any{"page": "/home", "event_id": "e1"} + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", body))) + require.Equal(t, tt.status, w.Code) + assert.Equal(t, tt.retryAfter, w.Header().Get("Retry-After")) + assert.True(t, dedup.Pending(dedupe.Key{Table: "clicks", ID: "e1"}), "left to lapse") + assert.Empty(t, dedup.Released) + + pub.Err = nil + w = httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", body))) + assert.Equal(t, http.StatusServiceUnavailable, w.Code) + assert.Equal(t, "30", w.Header().Get("Retry-After")) + assert.Empty(t, pub.Published()) + }) + } +} + +// A claimed record is published under its idempotency key; an un-deduped one +// carries none, so a producer's repeated ids are not dropped by the queue. +func TestIngest_Dedup_PublishCarriesTheIdempotencyKey(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := dedupHandler(t, pub, testutil.NewMockDeduplicator(), false) + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", + jsonLine(t, map[string]any{"page": "/a", "event_id": "e1"}), + jsonLine(t, map[string]any{"page": "/b"}), + ))) + require.Equal(t, http.StatusOK, w.Code) + msgs := pub.Published() + require.Len(t, msgs, 2) + + want := mq.Headers{} + mq.WithIdempotencyKey(dedupe.IdempotencyKey(testStore.Tenant(), dedupe.Key{Table: "clicks", ID: "e1"}))(want) + for k, v := range want { + assert.Equal(t, v, msgs[0].Headers[k]) + assert.NotContains(t, msgs[1].Headers, k) + } +} + +// A release that fails after a failed publish is only logged: the request +// answers with the publish's own error, and the id is left to lapse with its +// lease rather than being reported as a dedupe failure. The publish error must +// be definite (mq.ErrQueueFull) for a single-record window: an uncertain +// failure never releases its own record's claim (it is left to lapse +// instead), so there would be nothing to release. +func TestIngest_Dedup_FailedReleaseKeepsThePublishError(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull)} + dedup := testutil.NewMockDeduplicator() + dedup.ReleaseErr = errors.New("store unavailable") + h := dedupHandler(t, pub, dedup, false) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "e1"}))) + require.Equal(t, http.StatusServiceUnavailable, w.Code) + assert.NotContains(t, w.Body.String(), "dedupe", "the publish's error, not the release's") + assert.Len(t, dedup.Released, 1, "the release was attempted") + assert.True(t, dedup.Pending(dedupe.Key{Table: "clicks", ID: "e1"}), "left to lapse with its lease") +} + +// A batch whose publish fails part-way keeps what it published: the records +// before the failure are committed, the failing one is released, and a +// whole-batch retry reports the first as duplicates and publishes the rest. +func TestIngest_NDJSON_Dedup_PublishFailureMidBatch(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull), ErrAfter: 1} + dedup := testutil.NewMockDeduplicator() + h := dedupHandler(t, pub, dedup, false) + batch := func() *http.Request { + return ndjsonRequest(t, "clicks", + jsonLine(t, map[string]any{"page": "/a", "event_id": "e1"}), + jsonLine(t, map[string]any{"page": "/b", "event_id": "e2"}), + jsonLine(t, map[string]any{"page": "/c", "event_id": "e3"}), + ) + } + + w := httptest.NewRecorder() + h.Handle(w, withTenant(batch())) + require.Equal(t, http.StatusServiceUnavailable, w.Code) + require.Len(t, dedup.Released, 2, "the failing record and the rest of its window") + assert.Equal(t, dedupe.Key{Table: "clicks", ID: "e2"}, dedup.Released[0].Key) + assert.Equal(t, dedupe.Key{Table: "clicks", ID: "e3"}, dedup.Released[1].Key) + + pub.Err = nil + w = httptest.NewRecorder() + h.Handle(w, withTenant(batch())) + require.Equal(t, http.StatusOK, w.Code) + resp := decodeBatchResult(t, w) + assert.Equal(t, 1, resp.Duplicates, "e1 was published by the first attempt") + assert.Equal(t, 2, resp.Succeeded) + assert.Len(t, pub.Published(), 3, "every record exactly once") +} + +// An id another request holds answers 503 with the lease as Retry-After: +// that request's publish decides whether this record is a duplicate. +func TestIngest_Dedup_InFlight(t *testing.T) { + t.Parallel() + tests := []struct { + name string + lease time.Duration + retryAfter string + }{ + {"default lease", 0, "30"}, + {"configured lease rounds up", 4500 * time.Millisecond, "5"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + dedup.Hold(dedupe.Key{Table: "clicks", ID: "e1"}) + h := dedupHandler(t, pub, dedup, false) + h.DedupeLease = tt.lease + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "e1"}))) + assert.Equal(t, http.StatusServiceUnavailable, w.Code) + assert.Equal(t, tt.retryAfter, w.Header().Get("Retry-After")) + assert.Contains(t, w.Body.String(), "in flight") + assert.Empty(t, pub.Published()) + }) + } +} + +// A commit that fails after the publish does not fail the record: it is in +// the queue, and answering an error would invite a second copy. +func TestIngest_Dedup_CommitFailureStillSucceeds(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + dedup.CommitErr = errors.New("disk full") + h := dedupHandler(t, pub, dedup, false) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "e1"}))) + assert.Equal(t, http.StatusOK, w.Code) + assert.Len(t, pub.Published(), 1) +} + +// A dedupe backend error before the publish publishes nothing and fails the +// request, as before. +func TestIngest_Dedup_ReserveError(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + dedup.Err = errors.New("backend down") + h := dedupHandler(t, pub, dedup, false) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "e1"}))) + assert.Equal(t, http.StatusInternalServerError, w.Code) + assert.Contains(t, w.Body.String(), "dedupe failed") + assert.Empty(t, pub.Published()) +} + +// A Reserve error's log level depends on why it failed. A real backend +// failure (a live request context) stays ERROR, so an operator is paged. One +// caused by the request's own context ending (the client gone, or its +// deadline past) is not a backend problem and must log at DEBUG instead — an +// operator paging on ERROR logs would otherwise be woken by clients that +// simply went away. Not t.Parallel: it captures the process-wide default +// logger (logtest.Capture). Matched on the exact "level":"…","msg":"…" pair +// slog's JSON handler emits adjacently, not on the level alone — the package +// also logs an unrelated "debug: span started for ingest" line per request, +// which satisfies a bare `"level":"DEBUG"` check whether or not the Reserve +// line itself is DEBUG. +func TestIngest_Dedup_ReserveError_LogLevel(t *testing.T) { + tests := []struct { + name string + cancelContext bool + wantLine string + }{ + { + "live context: a real backend failure pages at ERROR", false, + `"level":"ERROR","msg":"dedupe reserve failed"`, + }, + { + "context ended: a client gone must not page, logs at DEBUG", true, + `"level":"DEBUG","msg":"dedupe reserve failed: request context ended"`, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + buf := logtest.Capture(t, slog.LevelDebug) + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + dedup.Err = errors.New("backend down") + h := dedupHandler(t, pub, dedup, false) + + ctx := context.Background() + if tt.cancelContext { + var cancel context.CancelFunc + ctx, cancel = context.WithCancel(ctx) + cancel() + } + req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "e1"}).WithContext(ctx) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(req)) + + assert.Contains(t, buf.String(), tt.wantLine) + assert.Empty(t, pub.Published()) + }) + } +} + +// #370: an explicit null id is a missing id — rejected under require_id, +// published un-deduped otherwise — never the one id "" that made every +// null record after the first a duplicate. +func TestIngest_Dedup_NullIDIsMissing(t *testing.T) { + t.Parallel() + nullID := func() string { return jsonLine(t, map[string]any{"page": "/a", "event_id": nil}) } + t.Run("require_id rejects", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := dedupHandler(t, pub, testutil.NewMockDeduplicator(), true) + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", nullID()))) + require.Equal(t, http.StatusOK, w.Code) + assert.Contains(t, resultAt(t, decodeBatchResult(t, w), 1).Error, "missing dedupe id field") + assert.Empty(t, pub.Published()) + }) + t.Run("otherwise publishes every one", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := dedupHandler(t, pub, testutil.NewMockDeduplicator(), false) + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", nullID(), nullID()))) + require.Equal(t, http.StatusOK, w.Code) + resp := decodeBatchResult(t, w) + assert.Equal(t, 2, resp.Succeeded) + assert.Equal(t, 0, resp.Duplicates) + assert.Len(t, pub.Published(), 2) + }) +} + +// #222: the key carries the table, so one id value in two tables is two ids. +func TestIngest_Dedup_KeyedByTable(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + h := dedupHandler(t, pub, dedup, false) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "e1"}))) + require.Equal(t, http.StatusOK, w.Code) + claims, err := dedup.Reserve(t.Context(), []dedupe.Key{{Table: "clicks", ID: "e1"}, {Table: "views", ID: "e1"}}, time.Second) + require.NoError(t, err) + assert.Equal(t, dedupe.Duplicate, claims[0].Status) + assert.Equal(t, dedupe.Claimed, claims[1].Status) +} + +// #390: concurrent requests carrying one id publish it once, over the real +// embedded store — the rest answer duplicate, or 503 while the winner is +// still publishing. +func TestIngest_Dedup_ConcurrentSameIDPublishesOnce(t *testing.T) { + t.Parallel() + store := dedupe.NewEmbedded(t.TempDir()).Tenant("acme") + require.NoError(t, store.Apply(true)) + t.Cleanup(func() { _ = store.Close() }) + pub := &testutil.MockPublisher{} + h := dedupHandler(t, pub, store, false) + + const n = 32 + codes := make([]int, n) + var wg sync.WaitGroup + for i := range n { + wg.Go(func() { + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "e1"}))) + codes[i] = w.Code + }) + } + wg.Wait() + assert.Len(t, pub.Published(), 1) + for _, c := range codes { + assert.Contains(t, []int{http.StatusOK, http.StatusServiceUnavailable}, c) + } +} diff --git a/internal/api/ingest_window_test.go b/internal/api/ingest_window_test.go new file mode 100644 index 000000000..1fd13fb65 --- /dev/null +++ b/internal/api/ingest_window_test.go @@ -0,0 +1,442 @@ +package api + +import ( + "context" + "errors" + "fmt" + "net/http" + "net/http/httptest" + "strings" + "sync" + "testing" + "time" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/mq" + "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/testutil" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// eventLines is n NDJSON clicks records with ids e1..en. +func eventLines(t *testing.T, n int) []string { + t.Helper() + lines := make([]string, n) + for i := range n { + lines[i] = jsonLine(t, map[string]any{"page": "/p", "event_id": fmt.Sprintf("e%d", i+1)}) + } + return lines +} + +func clickKey(i int) dedupe.Key { return dedupe.Key{Table: "clicks", ID: fmt.Sprintf("e%d", i)} } + +// A batch is reserved, published and committed a window at a time: one +// Reserve and one Commit per window, whatever the batch size. +func TestIngest_Windows_OneReserveAndCommitPerWindow(t *testing.T) { + t.Parallel() + for _, n := range []int{1, 255, 256, 257, 600} { + t.Run(fmt.Sprint(n), func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + h := dedupHandler(t, pub, dedup, false) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", eventLines(t, n)...))) + require.Equal(t, http.StatusOK, w.Code) + assert.Equal(t, n, decodeBatchResult(t, w).Succeeded) + windows := (n + ingestWindow - 1) / ingestWindow + assert.Equal(t, windows, dedup.Reserves) + assert.Equal(t, windows, dedup.Commits) + assert.Len(t, pub.Published(), n) + assert.True(t, dedup.Committed(clickKey(n))) + }) + } +} + +// A publish failing at record k settles its window: the records before k are +// committed, k is released when the queue refused it and left to lapse when +// the outcome is unknown, the rest of the window is released, and later +// windows are never reserved. A whole-batch retry after a refusal publishes +// every record exactly once. +func TestIngest_Windows_PublishFailureAtK(t *testing.T) { + t.Parallel() + const n = 600 + refused := fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull) + tests := []struct { + name string + k int + err error + status int + }{ + {"refused first record", 1, refused, http.StatusServiceUnavailable}, + {"refused mid first window", 100, refused, http.StatusServiceUnavailable}, + {"refused last of first window", 256, refused, http.StatusServiceUnavailable}, + {"refused first of second window", 257, refused, http.StatusServiceUnavailable}, + {"refused mid last window", 590, refused, http.StatusServiceUnavailable}, + {"uncertain mid first window", 100, context.DeadlineExceeded, http.StatusInternalServerError}, + {"uncertain mid second window", 400, context.DeadlineExceeded, http.StatusInternalServerError}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{Err: tt.err, ErrAfter: tt.k - 1} + dedup := testutil.NewMockDeduplicator() + h := dedupHandler(t, pub, dedup, false) + lines := eventLines(t, n) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", lines...))) + require.Equal(t, tt.status, w.Code) + assert.Len(t, pub.Published(), tt.k-1) + + windowEnd := min((tt.k-1)/ingestWindow*ingestWindow+ingestWindow, n) + definite := errors.Is(tt.err, mq.ErrQueueFull) + var released []dedupe.Key + for _, c := range dedup.Released { + released = append(released, c.Key) + } + var wantReleased []dedupe.Key + for i := tt.k; i <= windowEnd; i++ { + if i > tt.k || definite { + wantReleased = append(wantReleased, clickKey(i)) + } + } + assert.Equal(t, wantReleased, released) + if tt.k > 1 { + assert.True(t, dedup.Committed(clickKey(1))) + assert.True(t, dedup.Committed(clickKey(tt.k-1)), "published before the failure") + } + assert.False(t, dedup.Committed(clickKey(tt.k))) + assert.Equal(t, !definite, dedup.Pending(clickKey(tt.k)), "an uncertain publish leaves its claim to lapse") + if windowEnd < n { + assert.False(t, dedup.Pending(clickKey(windowEnd+1)), "a later window is never reserved") + } + if !definite { + return + } + + pub.Err = nil + w = httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", lines...))) + require.Equal(t, http.StatusOK, w.Code) + resp := decodeBatchResult(t, w) + assert.Equal(t, tt.k-1, resp.Duplicates) + assert.Equal(t, n-(tt.k-1), resp.Succeeded) + assert.Len(t, pub.Published(), n, "every record exactly once") + }) + } +} + +// A dedupe store that cannot answer now fails the request with 503 and a +// short Retry-After, which the SDK retries — not the 500 of a broken store. +// Earlier windows stay published and committed. +func TestIngest_Dedup_UnavailableIs503(t *testing.T) { + t.Parallel() + notOpen := dedupe.NewManaged(func() (dedupe.Deduplicator, error) { return nil, errors.New("disk gone") }) + require.Error(t, notOpen.Apply(true)) + throttled := testutil.NewMockDeduplicator() + throttled.Err = fmt.Errorf("%w: throttled", dedupe.ErrUnavailable) + secondWindow := testutil.NewMockDeduplicator() + secondWindow.Err, secondWindow.ErrAfter = throttled.Err, 1 + + tests := []struct { + name string + dedup dedupe.Deduplicator + n int + published int + }{ + {"store not open", notOpen, 1, 0}, + {"backend throttled", throttled, 3, 0}, + {"second window throttled", secondWindow, ingestWindow + 1, ingestWindow}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := dedupHandler(t, pub, tt.dedup, false) + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", eventLines(t, tt.n)...))) + assert.Equal(t, http.StatusServiceUnavailable, w.Code) + assert.Equal(t, "5", w.Header().Get("Retry-After")) + assert.Contains(t, w.Body.String(), "dedupe store unavailable") + assert.Len(t, pub.Published(), tt.published) + }) + } + t.Run("single object", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := dedupHandler(t, pub, throttled, false) + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/", "event_id": "e1"}))) + assert.Equal(t, http.StatusServiceUnavailable, w.Code) + assert.Equal(t, "5", w.Header().Get("Retry-After")) + assert.Empty(t, pub.Published()) + }) +} + +// One id held by another request stops its window before anything in it is +// published and gives back the window's other claims; windows before it stay +// committed. +func TestIngest_Windows_InFlightReleasesTheWindow(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + held := ingestWindow + 2 + dedup.Hold(clickKey(held)) + h := dedupHandler(t, pub, dedup, false) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", eventLines(t, ingestWindow+3)...))) + assert.Equal(t, http.StatusServiceUnavailable, w.Code) + assert.Equal(t, "30", w.Header().Get("Retry-After")) + assert.Len(t, pub.Published(), ingestWindow) + assert.True(t, dedup.Committed(clickKey(ingestWindow))) + for _, i := range []int{ingestWindow + 1, ingestWindow + 3} { + assert.False(t, dedup.Pending(clickKey(i)), "e%d released", i) + } + assert.True(t, dedup.Pending(clickKey(held)), "the other request's claim is untouched") +} + +// Rejects, duplicates and repeats keep their places in the results across +// windows, over both batch formats. +func TestIngest_Windows_OutcomesStayInOrder(t *testing.T) { + t.Parallel() + records := []map[string]any{ + {"page": "/a", "event_id": "e1"}, + {"page": "/b", "event_id": "e1"}, // repeat inside one window + {"page": "/c", "nope": 1}, // reject + {"page": "/d", "event_id": "e2"}, + {"page": "/e", "event_id": "e1"}, // repeat across windows + {"page": "/f"}, // no id: published un-deduped + } + requests := map[string]func() *http.Request{ + "ndjson": func() *http.Request { + lines := make([]string, len(records)) + for i, r := range records { + lines[i] = jsonLine(t, r) + } + return ndjsonRequest(t, "clicks", lines...) + }, + "json array": func() *http.Request { return ingestRequest(t, "clicks", records) }, + } + for name, req := range requests { + t.Run(name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + h := dedupHandler(t, pub, dedup, false) + h.window = 3 + + w := httptest.NewRecorder() + h.Handle(w, withTenant(req())) + require.Equal(t, http.StatusOK, w.Code) + resp := decodeBatchResult(t, w) + assert.Equal(t, []recordResult{ + {Index: 1, Ok: true}, + {Index: 2, Duplicate: true}, + {Index: 3, Error: resp.Results[2].Error}, + {Index: 4, Ok: true}, + {Index: 5, Duplicate: true}, + {Index: 6, Ok: true}, + }, resp.Results) + assert.NotEmpty(t, resp.Results[2].Error) + assert.Equal(t, 6, resp.Total) + assert.Len(t, pub.Published(), 3) + assert.Equal(t, 2, dedup.Reserves) + }) + } +} + +// The embedded queue must remember an idempotency key for at least two +// leases plus a second: the in-flight 503 of an uncertain publish sends the +// full lease as Retry-After, so a client that obeys it can republish up to +// ~2*lease after the original Reserve, and a claim's expiry can itself round +// up by up to a second (a DynamoDB backend, for one). Only the queue's +// duplicate window running at least that long guarantees it still drops the +// retry's second copy. +func TestIngest_DedupeLeaseFitsTheDuplicateWindow(t *testing.T) { + t.Parallel() + assert.LessOrEqual(t, 2*dedupe.DefaultLease+time.Second, mq.EmbeddedDuplicateWindow) +} + +// faultyPublisher publishes through a real broker and fails the calls fail +// picks: before sending (the queue refused it) or after (the outcome unknown +// to the caller, though the event is stored). +type faultyPublisher struct { + mq.Publisher + mu sync.Mutex + calls int + fail func(call int) (sendFirst bool, err error) +} + +func (p *faultyPublisher) Publish(ctx context.Context, topic mq.Topic, data []byte, opts ...mq.PublishOpt) error { + p.mu.Lock() + p.calls++ + sendFirst, err := p.fail(p.calls) + p.mu.Unlock() + if err == nil || sendFirst { + if pubErr := p.Publisher.Publish(ctx, topic, data, opts...); pubErr != nil { + return pubErr + } + } + return err +} + +// realPipeline is an ingest handler over the embedded broker and Pebble +// store, with pub's faults in front of the broker, and a count of the events +// in the tenant's queue. +func realPipeline(t *testing.T, fail func(call int) (bool, error)) (*IngestHandler, func() int) { + t.Helper() + broker, err := mq.NewEmbedded(storedir.New(t)) + require.NoError(t, err) + t.Cleanup(func() { _ = broker.Close() }) + require.NoError(t, broker.SetMaxBytes(t.Context(), testStore.Tenant(), 64<<20)) + store := dedupe.NewEmbedded(t.TempDir()).Tenant(testStore.Tenant()) + require.NoError(t, store.Apply(true)) + t.Cleanup(func() { _ = store.Close() }) + + h := dedupHandler(t, nil, store, false) + h.Publisher = &faultyPublisher{Publisher: broker, fail: fail} + count := func() int { + n := 0 + require.NoError(t, broker.ReplaySince(t.Context(), mq.Topic{Tenant: testStore.Tenant(), Table: "clicks"}, time.Time{}, + func([]byte) bool { n++; return true })) + return n + } + return h, count +} + +// #384 end to end: a publish the queue refused, then the client's retry, ends +// in exactly one event in the queue — and a later retry is a duplicate. +func TestIngest_Dedup_FailedPublishThenRetryIsOneEvent(t *testing.T) { + t.Parallel() + h, count := realPipeline(t, func(call int) (bool, error) { + if call == 1 { + return false, fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull) + } + return false, nil + }) + body := map[string]any{"page": "/home", "event_id": "e1"} + codes := make([]int, 3) + for i := range codes { + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", body))) + codes[i] = w.Code + if i == 2 { + assert.Contains(t, w.Body.String(), `"duplicate":true`) + } + } + assert.Equal(t, []int{http.StatusServiceUnavailable, http.StatusOK, http.StatusOK}, codes) + assert.Equal(t, 1, count()) +} + +// A publish that stored the event but reported a failure, then the client's +// retry: in-flight until the lease lapses, then republished under the same +// idempotency key, which the queue drops — one event, and the id committed. +func TestIngest_Dedup_UncertainPublishThenRetryIsOneEvent(t *testing.T) { + t.Parallel() + h, count := realPipeline(t, func(call int) (bool, error) { + if call == 1 { + return true, context.DeadlineExceeded + } + return false, nil + }) + h.DedupeLease = 2 * time.Second + lines := eventLines(t, 3) + send := func() *httptest.ResponseRecorder { + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", lines...))) + return w + } + + w := send() + require.Equal(t, http.StatusInternalServerError, w.Code) + w = send() + require.Equal(t, http.StatusServiceUnavailable, w.Code, "the uncertain claim is still held") + assert.Equal(t, "2", w.Header().Get("Retry-After")) + + var last *httptest.ResponseRecorder + require.Eventually(t, func() bool { + last = send() + return last.Code == http.StatusOK + }, 10*time.Second, 100*time.Millisecond) + assert.Equal(t, 3, decodeBatchResult(t, last).Succeeded, "the lapsed claim is claimed again and republished") + assert.Equal(t, 3, count(), "the republished e1 was dropped by the queue") + + w = send() + require.Equal(t, http.StatusOK, w.Code) + assert.Equal(t, 3, decodeBatchResult(t, w).Duplicates) +} + +// countingDedup counts the Commits that reach a store: on Pebble each is one +// fsync. +type countingDedup struct { + dedupe.Deduplicator + mu sync.Mutex + commits int +} + +func (c *countingDedup) Commit(ctx context.Context, claims []dedupe.Claim, retention time.Duration) error { + c.mu.Lock() + c.commits++ + c.mu.Unlock() + return c.Deduplicator.Commit(ctx, claims, retention) +} + +// pebbleBatchHandler is a handler over a real Pebble store behind a Commit +// counter, publishing to a mock queue. +func pebbleBatchHandler(tb testing.TB, window int) (*IngestHandler, *countingDedup) { + tb.Helper() + store := dedupe.NewEmbedded(tb.TempDir()).Tenant(testStore.Tenant()) + require.NoError(tb, store.Apply(true)) + tb.Cleanup(func() { _ = store.Close() }) + counted := &countingDedup{Deduplicator: store} + h := NewIngestHandler(fixedRegistry(testRegistry(tb)), &testutil.MockPublisher{}) + h.Dedup = staticDedup(counted) + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } + h.window = window + return h, counted +} + +// Windows cut the per-record fsyncs on Pebble: a 1,000-record batch commits in +// four syncs rather than a thousand. +func TestIngest_Windows_OneSyncPerWindowOnPebble(t *testing.T) { + t.Parallel() + h, counted := pebbleBatchHandler(t, 0) + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", eventLines(t, 1000)...))) + require.Equal(t, http.StatusOK, w.Code) + assert.Equal(t, 4, counted.commits) +} + +// BenchmarkIngest_DedupBatchOnPebble compares a 1,000-record batch committed +// per record (window 1, the pre-window behavior) with the default window. +func BenchmarkIngest_DedupBatchOnPebble(b *testing.B) { + for _, window := range []int{1, ingestWindow} { + b.Run(fmt.Sprintf("window=%d", window), func(b *testing.B) { + h, counted := pebbleBatchHandler(b, window) + var body strings.Builder + iter := 0 + for b.Loop() { + iter++ + body.Reset() + for i := range 1000 { + fmt.Fprintf(&body, `{"page":"/p","event_id":"%d-%d"}`+"\n", iter, i) + } + req := httptest.NewRequestWithContext(context.Background(), http.MethodPost, "/v1/ingest?table=clicks", strings.NewReader(body.String())) + req.Header.Set("Content-Type", "application/x-ndjson") + w := httptest.NewRecorder() + h.Handle(w, withTenant(req)) + if w.Code != http.StatusOK { + b.Fatalf("status %d", w.Code) + } + } + b.ReportMetric(float64(counted.commits)/float64(iter), "syncs/op") + }) + } +} diff --git a/internal/api/settings_test.go b/internal/api/settings_test.go index 9e823ceaa..462cbc8e6 100644 --- a/internal/api/settings_test.go +++ b/internal/api/settings_test.go @@ -16,10 +16,10 @@ import ( "github.com/stretchr/testify/require" ) -// fullConfig is a complete config.json (every key is required) with the +// fullConfig is a complete config.json (every key set) with the // given query.default_max_rows. func fullConfig(maxRows int) string { - return fmt.Sprintf(`{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": %d, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, maxRows) + return fmt.Sprintf(`{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": %d, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, maxRows) } // writeSettingsFixture materializes a minimal valid settings directory whose diff --git a/internal/app/app.go b/internal/app/app.go index 939ff29b0..fc028cb76 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -187,7 +187,7 @@ func New(ctx context.Context, opts Options) (app *App, err error) { } if apiRole { a.wireDiscovery(ctx) - if err := a.wireDedupe(); err != nil { + if err := a.wireDedupe(ctx); err != nil { return nil, err } } diff --git a/internal/app/app_test.go b/internal/app/app_test.go index 945f7eb96..80a2e8c58 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -31,11 +31,13 @@ import ( "github.com/Wave-RF/WaveHouse/internal/config" "github.com/Wave-RF/WaveHouse/internal/coord" "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/dedupe/dedupetest" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/settings" "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" ) // None of these tests run in parallel: New installs a process-wide default @@ -95,7 +97,7 @@ func writeSettings(t *testing.T, patch map[string]any) string { func testConfig(t *testing.T, settingsDir string) *config.Config { t.Helper() return &config.Config{ - DataDir: t.TempDir(), + DataDir: storedir.New(t), Server: config.Server{Port: closedPort(t), ShutdownTimeout: 2}, MQ: config.MQ{Backend: config.MQEmbedded}, Cache: config.Cache{Backend: config.CacheLocal, L1MaxCost: 1 << 20}, @@ -213,7 +215,7 @@ func TestNew_DedupeFollowsSettings(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { dir := writeSettings(t, map[string]any{"dedupe": map[string]any{ - "enabled": tt.enabled, "id_field": "event_id", "require_id": false, "tables": map[string]any{}, + "enabled": tt.enabled, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}, }}) cfg := testConfig(t, dir) a := newApp(t, cfg, Options{}) @@ -246,7 +248,7 @@ func TestReload_DrivesTheRegisteredHooks(t *testing.T) { require.Equal(t, int64(1<<30), a.mq.MaxBytes(tenant.Default)) rewriteSettings(t, dir, map[string]any{ - "dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}, + "dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}}, "mq": map[string]any{"max_bytes_gb": 2}, }) _, adopted := a.tenants.Reload("test") @@ -410,7 +412,7 @@ func TestNew_NestedWithoutAnOperatorKeyWarnsTheOpsTreeIsClosed(t *testing.T) { // request, so a lost 0 folder is felt at once on the routes that read tenant // 0's list. func TestReload_NestedHooksFollowEachTenant(t *testing.T) { - dedupeOn := map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}} + dedupeOn := map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}} grown := map[string]any{"dedupe": dedupeOn, "mq": map[string]any{"max_bytes_gb": 2}} root := writeNestedSettings(t, map[string]map[string]any{ "0": {"mq": map[string]any{"max_bytes_gb": 1}}, @@ -483,7 +485,7 @@ func TestReload_NestedHooksFollowEachTenant(t *testing.T) { // reopened over the same seen ids when the folder is back. The instance is // open while some tenant's store is, and Close releases it. func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { - dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}} + dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}}} root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn, "globex": nil, "broken": invalidQuery}) cfg := testConfig(t, root) a := newApp(t, cfg, Options{}) @@ -496,14 +498,14 @@ func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { for _, id := range []string{"acme", "globex", "broken"} { assert.NoDirExists(t, filepath.Join(cfg.DataDir, id), "and no directory of a tenant's own") } - dup, err := acme.CheckAndMark(ctx, "e1") + dup, err := dedupetest.Mark(ctx, acme, eventKey) require.NoError(t, err) assert.False(t, dup) rewriteSettings(t, filepath.Join(root, "globex"), dedupeOn) a.tenants.Reload("test") assert.True(t, globex.Open(), "globex's reload opens globex's store") - dup, err = globex.CheckAndMark(ctx, "e1") + dup, err = dedupetest.Mark(ctx, globex, eventKey) require.NoError(t, err) assert.False(t, dup, "an id acme has seen is new to globex") @@ -533,7 +535,7 @@ func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { a.tenants.Reload("test") restored := a.dedup.For("acme") assert.True(t, restored.Open()) - dup, err = restored.CheckAndMark(ctx, "e1") + dup, err = dedupetest.Mark(ctx, restored, eventKey) require.NoError(t, err) assert.True(t, dup, "an id seen before the folder was removed is still a duplicate") @@ -567,10 +569,10 @@ func TestNew_RefusesALayerWithoutABackend(t *testing.T) { // A Pebble instance that cannot open follows the registry's own rule for the // shape: a flat directory refuses boot, like every other store, and a nested // one fails closed for every tenant with dedupe on, since they share the -// instance — their ingest answers 500 until a reload or a restart opens it — +// instance — their ingest answers 503 until a reload or a restart opens it — // while the process, and every tenant with dedupe off, carries on. func TestNew_DedupeOpenFailure(t *testing.T) { - dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}} + dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}}} // A regular file where the instance's directory should be is what Pebble // refuses to open. block := func(t *testing.T, dataDir string) { @@ -592,10 +594,10 @@ func TestNew_DedupeOpenFailure(t *testing.T) { for _, id := range []tenant.ID{"acme", "globex"} { store := a.dedup.For(id) assert.False(t, store.Open()) - _, err := store.CheckAndMark(t.Context(), "e1") + _, err := dedupetest.Mark(t.Context(), store, eventKey) require.ErrorIs(t, err, dedupe.ErrUnavailable, "%s: switched on but not open, so its ingest fails closed", id) } - _, err := a.dedup.For("initech").CheckAndMark(t.Context(), "e1") + _, err := dedupetest.Mark(t.Context(), a.dedup.For("initech"), eventKey) require.ErrorIs(t, err, dedupe.ErrDisabled, "a tenant with dedupe off is as it would be anyway") }) } @@ -766,6 +768,7 @@ func redisTestConfig(t *testing.T, settingsDir, addr string) *config.Config { Timeout: 100 * time.Millisecond, DialTimeout: 200 * time.Millisecond, MaxValueBytes: 1 << 20, CompressMinBytes: 1 << 10, VersionTTL: time.Hour, }} + cfg.Dedupe.Lease, cfg.Dedupe.ReserveConcurrency = 30*time.Second, 64 require.NoError(t, cfg.Validate()) return cfg } @@ -1026,7 +1029,7 @@ func analystPipe(t *testing.T, dir string) { func TestNew_LateBootFailureReleasesEverything(t *testing.T) { guardGlobals(t) dir := writeSettings(t, map[string]any{"dedupe": map[string]any{ - "enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}, + "enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}, }}) cfg := testConfig(t, dir) natsDir := filepath.Join(cfg.DataDir, "nats") @@ -1656,7 +1659,7 @@ func TestReload_CeilingRefusesAThirdTupleThenOpensIt(t *testing.T) { func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { jwks, _, fetches := jwksServer(t, "acme-1") acmeSettings := authPatch(jwks.URL) - acmeSettings["dedupe"] = map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}} + acmeSettings["dedupe"] = map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}} root := writeNestedSettings(t, map[string]map[string]any{"acme": acmeSettings, "globex": nil}) a := newApp(t, testConfig(t, root), Options{}) acme, acmeRegistry, acmeDedup := a.pools.For("acme"), a.discoveries.For("acme"), a.dedup.For("acme") @@ -1679,7 +1682,7 @@ func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { a.Handler().ServeHTTP(rec, req) return fmt.Sprintf("%d %s", rec.Code, rec.Body.String()) } - dup, err := acmeDedup.CheckAndMark(t.Context(), "e1") + dup, err := dedupetest.Mark(t.Context(), acmeDedup, eventKey) require.NoError(t, err) require.False(t, dup) require.Eventually(t, func() bool { return fetches.Load() > 0 }, 5*time.Second, 10*time.Millisecond, "acme's key set is fetched off the boot path") @@ -1723,7 +1726,7 @@ func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { assert.NotNil(t, a.discoveries.For("acme")) assert.NotSame(t, acmeRegistry, a.discoveries.For("acme"), "and a fresh registry") assert.Eventually(t, func() bool { return fetches.Load() > fetched }, 5*time.Second, 10*time.Millisecond, "and a fresh verifier, fetching the key set again") - dup, err = a.dedup.For("acme").CheckAndMark(t.Context(), "e1") + dup, err = dedupetest.Mark(t.Context(), a.dedup.For("acme"), eventKey) require.NoError(t, err) assert.True(t, dup, "an id acme sent before the removal is still a duplicate") } @@ -1815,3 +1818,6 @@ func TestClose_StopsTheDiscoveryLoops(t *testing.T) { assert.Nil(t, a.discoveries.For("acme")) assert.Nil(t, a.pools.For("acme")) } + +// eventKey is the one dedupe key the tenant-lifecycle tests mark. +var eventKey = dedupe.Key{Table: "events", ID: "e1"} diff --git a/internal/app/dedupe_dynamodb_test.go b/internal/app/dedupe_dynamodb_test.go new file mode 100644 index 000000000..e9dd5290c --- /dev/null +++ b/internal/app/dedupe_dynamodb_test.go @@ -0,0 +1,406 @@ +package app + +import ( + "bytes" + "context" + "io" + "log/slog" + "net" + "net/http" + "net/http/httptest" + "path/filepath" + "strings" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/config" + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/dedupe/dedupetest" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// fakeDynamo answers the DynamoDB JSON protocol for one table, enough for +// boot's check, the dev create path, and a claim and its commit. Whether the +// table exists, whether every call is throttled, and whether the endpoint +// hangs (every call, or one op alone), are the test's to switch. +type fakeDynamo struct { + mu sync.Mutex + exists bool + throttles bool + hangs bool + hangOn string // hang calls of this op alone, once set; "" hangs none this way + calls []string +} + +func (f *fakeDynamo) setThrottles(v bool) { + f.mu.Lock() + defer f.mu.Unlock() + f.throttles = v +} + +func (f *fakeDynamo) setExists(v bool) { + f.mu.Lock() + defer f.mu.Unlock() + f.exists = v +} + +func (f *fakeDynamo) setHangs(v bool) { + f.mu.Lock() + defer f.mu.Unlock() + f.hangs = v +} + +func (f *fakeDynamo) setHangOn(op string) { + f.mu.Lock() + defer f.mu.Unlock() + f.hangOn = op +} + +func (f *fakeDynamo) called(op string) bool { return f.count(op) > 0 } + +func (f *fakeDynamo) count(op string) int { + f.mu.Lock() + defer f.mu.Unlock() + n := 0 + for _, c := range f.calls { + if c == op { + n++ + } + } + return n +} + +func (f *fakeDynamo) ServeHTTP(w http.ResponseWriter, r *http.Request) { + // Drained before any hang below: with the body unread, an SDK write + // deadline or the client giving up never reaches this handler, since the + // connection looks like it's still waiting for us to consume it. + _, _ = io.Copy(io.Discard, r.Body) + _, op, _ := strings.Cut(r.Header.Get("X-Amz-Target"), ".") + f.mu.Lock() + f.calls = append(f.calls, op) + if op == "CreateTable" { + f.exists = true + } + exists, throttled, hang := f.exists, f.throttles, f.hangs || op == f.hangOn + f.mu.Unlock() + if hang { + <-r.Context().Done() + return + } + w.Header().Set("Content-Type", "application/x-amz-json-1.0") + if throttled { + w.WriteHeader(http.StatusBadRequest) + _, _ = io.WriteString(w, `{"__type":"com.amazonaws.dynamodb.v20120810#ThrottlingException","message":"Rate exceeded"}`) + return + } + if !exists { + w.WriteHeader(http.StatusBadRequest) + _, _ = io.WriteString(w, `{"__type":"com.amazonaws.dynamodb.v20120810#ResourceNotFoundException","message":"Requested resource not found"}`) + return + } + body := `{}` + switch op { + case "DescribeTable", "CreateTable": + body = `{"Table":{"TableName":"dedupe","TableStatus":"ACTIVE",` + + `"KeySchema":[{"AttributeName":"pk","KeyType":"HASH"}],` + + `"AttributeDefinitions":[{"AttributeName":"pk","AttributeType":"S"}]}}` + case "DescribeTimeToLive": + body = `{"TimeToLiveDescription":{"AttributeName":"ex","TimeToLiveStatus":"ENABLED"}}` + case "BatchWriteItem": + body = `{"UnprocessedItems":{}}` + } + _, _ = io.WriteString(w, body) +} + +// dynamoConfig points cfg's dedupe at a fake table, with credentials from the +// environment as the SDK's default chain reads them — and nothing from the +// developer's own AWS files. +func dynamoConfig(t *testing.T, cfg *config.Config, exists bool) *fakeDynamo { + t.Helper() + fake := &fakeDynamo{exists: exists} + srv := httptest.NewServer(fake) + t.Cleanup(srv.Close) + none := filepath.Join(t.TempDir(), "none") + for k, v := range map[string]string{ + "AWS_ACCESS_KEY_ID": "local", "AWS_SECRET_ACCESS_KEY": "local", "AWS_SESSION_TOKEN": "", + "AWS_PROFILE": "", "AWS_CONFIG_FILE": none, "AWS_SHARED_CREDENTIALS_FILE": none, + "AWS_EC2_METADATA_DISABLED": "true", + } { + t.Setenv(k, v) + } + cfg.Dedupe = config.Dedupe{Backend: config.DedupeDynamoDB, DynamoDB: config.DedupeDynamoDBConfig{ + Table: "dedupe", Region: "us-east-1", Endpoint: srv.URL, MaxAttempts: 1, + }} + return fake +} + +var dedupeOn = map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}} + +func TestNew_DynamoDBDedupe(t *testing.T) { + cfg := testConfig(t, writeSettings(t, dedupeOn)) + fake := dynamoConfig(t, cfg, true) + a := newApp(t, cfg, Options{}) + + assert.True(t, fake.called("DescribeTable"), "boot checks the table") + assert.False(t, fake.called("CreateTable"), "and never creates it without create_table") + store := a.dedup.For(tenant.Default) + require.True(t, store.Open()) + dup, err := dedupetest.Mark(t.Context(), store, eventKey) + require.NoError(t, err) + assert.False(t, dup) + assert.True(t, fake.called("PutItem"), "the claim went to the table") + assert.True(t, fake.called("BatchWriteItem"), "and so did its commit") + assert.Nil(t, a.dedupeStats, "no Pebble instance, so no Pebble gauges") + assert.NoDirExists(t, filepath.Join(cfg.DataDir, "pebble")) +} + +func TestNew_DynamoDBDedupeCreatesTheTableOnlyWhenAsked(t *testing.T) { + cfg := testConfig(t, writeSettings(t, dedupeOn)) + fake := dynamoConfig(t, cfg, false) + cfg.Dedupe.DynamoDB.CreateTable = true + a := newApp(t, cfg, Options{}) + assert.True(t, fake.called("CreateTable")) + assert.True(t, fake.called("UpdateTimeToLive")) + assert.True(t, a.dedup.For(tenant.Default).Open()) +} + +// A misconfigured table refuses boot only over a flat directory in which a +// tenant has dedupe on; every other failure boots and fails closed. +func TestNew_DynamoDBDedupeTableMissing(t *testing.T) { + t.Run("flat with dedupe on refuses boot", func(t *testing.T) { + guardGlobals(t) + cfg := testConfig(t, writeSettings(t, dedupeOn)) + dynamoConfig(t, cfg, false) + _, err := New(t.Context(), Options{Config: cfg}) + require.ErrorContains(t, err, "dedupe open") + require.ErrorContains(t, err, "ResourceNotFoundException") + require.NotErrorIs(t, err, dedupe.ErrUnavailable) + }) + t.Run("flat with dedupe off boots, and fails closed once it is on", func(t *testing.T) { + dir := writeSettings(t, nil) + cfg := testConfig(t, dir) + dynamoConfig(t, cfg, false) + logs := bootLogged(t) + a, err := New(t.Context(), Options{Config: cfg}) + require.NoError(t, err) + t.Cleanup(func() { assert.NoError(t, a.Close(context.Background())) }) + assert.Contains(t, logs.String(), `level=ERROR msg="dedupe: dynamodb table is misconfigured`) + + store := a.dedup.For(tenant.Default) + _, err = dedupetest.Mark(t.Context(), store, eventKey) + require.ErrorIs(t, err, dedupe.ErrDisabled) + rewriteSettings(t, dir, dedupeOn) + a.tenants.Reload("test") + _, err = dedupetest.Mark(t.Context(), store, eventKey) + require.ErrorIs(t, err, dedupe.ErrUnavailable, "switched on by a reload while the table is missing: closed, not un-deduped") + }) + t.Run("nested fails closed", func(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn, "globex": nil}) + cfg := testConfig(t, root) + dynamoConfig(t, cfg, false) + a := newApp(t, cfg, Options{}) + + acme := a.dedup.For("acme") + assert.False(t, acme.Open()) + _, err := dedupetest.Mark(t.Context(), acme, eventKey) + require.ErrorIs(t, err, dedupe.ErrUnavailable, "switched on, table missing: ingest fails closed") + _, err = dedupetest.Mark(t.Context(), a.dedup.For("globex"), eventKey) + require.ErrorIs(t, err, dedupe.ErrDisabled) + }) +} + +// A transient failure (a throttle) never refuses boot, even over a flat +// directory with dedupe on: the tenant fails closed until the background +// retry's check passes. +func TestRun_DynamoDBDedupeFlatThrottledRecovers(t *testing.T) { + cfg := testConfig(t, writeSettings(t, dedupeOn)) + fake := dynamoConfig(t, cfg, true) + fake.setThrottles(true) + var lc net.ListenConfig + ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") + require.NoError(t, err) + a := newApp(t, cfg, Options{Listener: ln}) + store := a.dedup.For(tenant.Default) + require.False(t, store.Open()) + _, err = dedupetest.Mark(t.Context(), store, eventKey) + require.ErrorIs(t, err, dedupe.ErrUnavailable, "switched on, table throttled: ingest fails closed") + + _, stop := runApp(t, a, ln) + fake.setThrottles(false) + require.Eventually(t, store.Open, 10*time.Second, 50*time.Millisecond, "the retry opened the store") + _, err = dedupetest.Mark(context.Background(), store, eventKey) + require.NoError(t, err) + require.NoError(t, stop()) +} + +// With create_table on, an endpoint that fails transiently (dynamodb-local +// still starting) boots too, and the retry creates the table once it answers. +func TestRun_DynamoDBDedupeFlatCreateTableRetries(t *testing.T) { + cfg := testConfig(t, writeSettings(t, dedupeOn)) + fake := dynamoConfig(t, cfg, false) + cfg.Dedupe.DynamoDB.CreateTable = true + fake.setThrottles(true) + var lc net.ListenConfig + ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") + require.NoError(t, err) + a := newApp(t, cfg, Options{Listener: ln}) + store := a.dedup.For(tenant.Default) + require.False(t, store.Open()) + + _, stop := runApp(t, a, ln) + fake.setThrottles(false) + require.Eventually(t, store.Open, 10*time.Second, 50*time.Millisecond, "the retry created the table and opened the store") + assert.True(t, fake.called("CreateTable")) + require.NoError(t, stop()) +} + +// bootLogged sends the default logger to a buffer for the rest of the test, +// for a boot that logs what it tolerated. +func bootLogged(t *testing.T) *lockedBuffer { + t.Helper() + guardGlobals(t) + buf := &lockedBuffer{} + slog.SetDefault(slog.New(slog.NewTextHandler(buf, nil))) + return buf +} + +// lockedBuffer is a bytes.Buffer safe for the background retry's logging. +type lockedBuffer struct { + mu sync.Mutex + buf bytes.Buffer +} + +func (b *lockedBuffer) Write(p []byte) (int, error) { + b.mu.Lock() + defer b.mu.Unlock() + return b.buf.Write(p) +} + +func (b *lockedBuffer) String() string { + b.mu.Lock() + defer b.mu.Unlock() + return b.buf.String() +} + +// The reload hook runs under the lock that serializes reloads, so it never +// calls DynamoDB: against a table that hangs, a reload returns at once, and a +// tenant it switches on fails closed rather than publishing un-deduped. +func TestReload_DynamoDBDedupeMakesNoTableCall(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": nil}) + cfg := testConfig(t, root) + fake := dynamoConfig(t, cfg, false) + a := newApp(t, cfg, Options{}) + fake.setHangs(true) + before := fake.count("DescribeTable") + + rewriteSettings(t, filepath.Join(root, "acme"), dedupeOn) + start := time.Now() + a.tenants.Reload("test") + assert.Less(t, time.Since(start), time.Second, "a check would wait out its 2.5s deadline") + assert.Equal(t, before, fake.count("DescribeTable"), "the reload made no table call") + + acme := a.dedup.For("acme") + assert.False(t, acme.Open()) + _, err := dedupetest.Mark(t.Context(), acme, eventKey) + require.ErrorIs(t, err, dedupe.ErrUnavailable, "switched on while the table fails: closed, not ErrDisabled") +} + +// A reload wakes the background retry rather than running the check itself. +// The retry's first timed attempt is a second after it starts, and a timer +// never fires early, so an open sooner than that is the reload's doing. +func TestRun_DynamoDBDedupeReloadWakesTheRetry(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn}) + cfg := testConfig(t, root) + fake := dynamoConfig(t, cfg, false) + var lc net.ListenConfig + ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") + require.NoError(t, err) + a := newApp(t, cfg, Options{Listener: ln}) + acme := a.dedup.For("acme") + require.False(t, acme.Open()) + + start := time.Now() + _, stop := runApp(t, a, ln) + fake.setExists(true) + a.tenants.Reload("test") + require.Eventually(t, acme.Open, 5*time.Second, 5*time.Millisecond) + assert.Less(t, time.Since(start), time.Second, "opened before the first timed retry") + _, err = dedupetest.Mark(context.Background(), acme, eventKey) + require.NoError(t, err) + require.NoError(t, stop()) +} + +// A nested directory has no watcher, so a table that comes good is picked up +// by the background retry, not only by a reload someone has to send. +func TestRun_DynamoDBDedupeRetriesTheTableCheck(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn}) + cfg := testConfig(t, root) + fake := dynamoConfig(t, cfg, false) + var lc net.ListenConfig + ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") + require.NoError(t, err) + a := newApp(t, cfg, Options{Listener: ln}) + acme := a.dedup.For("acme") + require.False(t, acme.Open()) + + _, stop := runApp(t, a, ln) + fake.setExists(true) + require.Eventually(t, acme.Open, 10*time.Second, 50*time.Millisecond, "the retry opened the store without a reload") + require.NoError(t, stop()) +} + +// No region anywhere is a certain config error: refused at boot in either +// shape rather than failing every check afterwards. +func TestNew_DynamoDBDedupeRefusesNoRegion(t *testing.T) { + for name, dir := range map[string]func(*testing.T) string{ + "flat": func(t *testing.T) string { return writeSettings(t, dedupeOn) }, + "nested": func(t *testing.T) string { return writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn}) }, + } { + t.Run(name, func(t *testing.T) { + guardGlobals(t) + cfg := testConfig(t, dir(t)) + dynamoConfig(t, cfg, true) + cfg.Dedupe.DynamoDB.Region = "" + t.Setenv("AWS_REGION", "") + t.Setenv("AWS_DEFAULT_REGION", "") + _, err := New(t.Context(), Options{Config: cfg}) + require.ErrorContains(t, err, "dynamodb region is not set") + }) + } +} + +// A reload must not wait behind a tenant's own in-flight DynamoDB call when +// nothing changes for that tenant: Managed.Apply's no-op fast path settles +// under a read lock alone, so it never contends with a Commit already +// holding one and returns long before the commit does. +func TestReload_DynamoDBDedupeDoesNotWaitOnInFlightCommit(t *testing.T) { + cfg := testConfig(t, writeSettings(t, dedupeOn)) + fake := dynamoConfig(t, cfg, true) + fake.setHangOn("BatchWriteItem") + a := newApp(t, cfg, Options{}) + + store := a.dedup.For(tenant.Default) + require.True(t, store.Open()) + claims, err := store.Reserve(context.Background(), []dedupe.Key{eventKey}, time.Minute) + require.NoError(t, err) + require.Equal(t, dedupe.Claimed, claims[0].Status) + + commitCtx, cancelCommit := context.WithCancel(context.Background()) + defer cancelCommit() + commitDone := make(chan error, 1) + go func() { commitDone <- store.Commit(commitCtx, claims, 0) }() + require.Eventually(t, func() bool { return fake.called("BatchWriteItem") }, time.Second, time.Millisecond, + "commit reached the table and is now hanging on it") + + start := time.Now() + a.tenants.Reload("test") + assert.Less(t, time.Since(start), 500*time.Millisecond, + "a reload that changes nothing for this tenant waited on its in-flight commit") + + cancelCommit() + <-commitDone // let the hung call finish (canceled) before the app closes +} diff --git a/internal/app/roles_test.go b/internal/app/roles_test.go index 48fb0fa12..c50eb9a32 100644 --- a/internal/app/roles_test.go +++ b/internal/app/roles_test.go @@ -17,6 +17,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/config" "github.com/Wave-RF/WaveHouse/internal/coord" "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" ) // Each role wires its own components and nothing else; the settings registry, @@ -109,7 +110,7 @@ func TestNew_OpsOnlyRouter(t *testing.T) { sweeperCfg := *cfg sweeperCfg.Roles = []config.Role{config.RoleSweeper} // Its own store: full's embedded JetStream is still open on cfg.DataDir. - sweeperCfg.DataDir = t.TempDir() + sweeperCfg.DataDir = storedir.New(t) a := newApp(t, &sweeperCfg, Options{}) for _, path := range []string{"/livez", "/readyz", "/healthz", "/version"} { diff --git a/internal/app/wire.go b/internal/app/wire.go index d48b76aec..a18ab5b45 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -53,7 +53,8 @@ func withoutContext(release func() error) func(context.Context) error { // configuration (dedupe, dlq, query, schema, stream, cors — see // settings.TenantConfig). Required: config.Validate already rejected an // empty settings.dir, and an invalid directory refuses boot. The binary -// carries no compiled defaults; `wavehouse bootstrap` writes the seed. A +// carries no compiled defaults but a missing dedupe.retention ("0"); +// `wavehouse bootstrap` writes the seed. A // *reload* of an invalid directory merely keeps the previous snapshot. A // nested directory (one folder per tenant, #583) fails closed per tenant // instead, at boot and on reload alike: see settings.Registry. @@ -464,10 +465,12 @@ func (a *App) wireDiscovery(ctx context.Context) { // wireDedupe builds the dedupe stores — the one place the implementation is // chosen. -func (a *App) wireDedupe() error { +func (a *App) wireDedupe(ctx context.Context) error { switch b := a.cfg.Dedupe.Backend; b { case config.DedupePebble: return a.wirePebbleDedupe() + case config.DedupeDynamoDB: + return a.wireDynamoDedupe(ctx) default: return unreachableBackend("dedupe.backend", b) } @@ -486,8 +489,9 @@ func (a *App) wireDedupe() error { // still closed — either the hook sees it or the boot apply reads it. An // instance that cannot open follows the registry's own rule for the shape: // flat refuses boot, like every other store, and on reload logs and leaves -// the store closed — ingest then fails closed (500 "dedupe failed") rather -// than silently publishing un-deduped, since the files asked for dedupe; +// the store closed — ingest then fails closed (503 "dedupe store +// unavailable", Retry-After: 5) rather than silently publishing un-deduped, +// since the files asked for dedupe; // nested fails closed the same way at boot too, for every tenant with // dedupe on, the next reload retrying, so it never costs the process. func (a *App) wirePebbleDedupe() error { @@ -532,6 +536,11 @@ func (a *App) wirePebbleDedupe() error { return nil } +// wireDynamoDedupe (dedupe.backend: dynamodb) lives in wire_dynamodb.go, +// excluded from the e2e coverage gate alongside internal/dedupe/dynamodb.go +// (see .testcoverage.yml): the e2e binary always runs Pebble dedupe, so +// nothing there exercises it. wireDedupe above still switches on it. + // wireMQ starts the MQ — the one place the implementation is chosen; // everything after it sees mq.Broker. func (a *App) wireMQ(ctx context.Context) error { @@ -967,6 +976,7 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { ingestHandler.PolicySource = (*settings.Store).Policy ingestHandler.Dedup = func(s *settings.Store) dedupe.Deduplicator { return a.dedup.For(s.Tenant()) } ingestHandler.DedupeSettings = (*settings.Store).DedupeFor + ingestHandler.DedupeLease = a.cfg.Dedupe.Lease // Readiness pings every open pool at once and is ready at the first // answer: one tenant's ClickHouse outage is not the process's. diff --git a/internal/app/wire_dynamodb.go b/internal/app/wire_dynamodb.go new file mode 100644 index 000000000..68f68ce45 --- /dev/null +++ b/internal/app/wire_dynamodb.go @@ -0,0 +1,143 @@ +package app + +import ( + "context" + "errors" + "fmt" + "log/slog" + "sync" + "time" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// errDynamoUnchecked is a store's open before the first table check has run. +var errDynamoUnchecked = errors.New("dedupe: dynamodb table not checked yet") + +// wireDynamoDedupe builds the dedupe stores over one DynamoDB table that +// every tenant and every process shares (dedupe.Dynamo), so a tenant's store +// opens for free once the table has passed its check. Boot checks it (after +// creating it, with create_table on dynamodb-local) whether or not any tenant +// has dedupe on, and never creates it otherwise. Boot is refused only when the +// table is misconfigured (a failure that is not ErrUnavailable: missing, the +// wrong key schema, access denied) over a flat directory in which a tenant +// has dedupe on. Otherwise — a transient failure, a nested directory, or no +// tenant deduping yet — the process boots with every switched-on store +// closed, so its ingest fails closed, and the check is retried in the +// background, with backoff, until it passes: a remote table's failure is +// often brief, a nested directory has no watcher to reload it, and a fixed +// table is picked up without a restart. +// The check is network I/O, so the AfterAdopt hook never runs it: the hook +// holds the lock that serializes reloads. It applies every store against the +// last check's result and wakes the retry, so a reload still retries at once. +func (a *App) wireDynamoDedupe(ctx context.Context) error { + c := a.cfg.Dedupe.DynamoDB + d, err := dedupe.NewDynamo(ctx, dedupe.DynamoConfig{ + Table: c.Table, Region: c.Region, Endpoint: c.Endpoint, + Timeout: c.Timeout, MaxAttempts: c.MaxAttempts, RetryMode: c.RetryMode, + ReserveConcurrency: a.cfg.Dedupe.ReserveConcurrency, + }) + if err != nil { + return err + } + var mu sync.Mutex + state := errDynamoUnchecked // nil once the table has passed, for good + ready := func() error { + mu.Lock() + defer mu.Unlock() + return state + } + // check is only ever run by boot, then by the retry loop, one at a time. + check := func(ctx context.Context) error { + var err error + if c.CreateTable { + err = d.CreateTable(ctx) + } + if err == nil { + err = d.Check(ctx) + } + mu.Lock() + defer mu.Unlock() + if state != nil { + state = err + } + return state + } + stores := dedupe.NewStores(dedupe.Factory(d.Tenant).Gated(ready)) + a.dedup = stores + a.add(component{name: "dedupe", close: withoutContext(stores.Close)}) + var reconciling sync.Mutex // the hook and the retry loop both apply + apply := func() { + reconciling.Lock() + defer reconciling.Unlock() + if err := stores.Retain(a.served); err != nil { + slog.Error("dedupe store close failed", "error", err) + } + for id, store := range a.tenants.All() { + m := stores.For(id) + enabled := store.DedupeEnabled() + wasOpen := m.Open() + // The one failure an open has is the check's, logged where it ran. + _ = m.Apply(enabled) + if m.Open() != wasOpen { + slog.Info("dedupe store reconciled with settings", "tenant", id, "enabled", enabled) + } + } + } + retry := make(chan struct{}, 1) + a.tenants.AfterAdopt(func([]tenant.ID) { + apply() + if ready() != nil { + select { + case retry <- struct{}{}: + default: // a retry is already due + } + } + }) + if err := check(ctx); err != nil { + misconfigured := !errors.Is(err, dedupe.ErrUnavailable) + if misconfigured && !a.tenants.Nested() && a.anyDedupeEnabled() { + return fmt.Errorf("dedupe open: %w", err) + } + if misconfigured { + slog.Error("dedupe: dynamodb table is misconfigured; ingest with dedupe on fails closed until it is fixed", + "table", c.Table, "error", err) + } else { + slog.Error("dedupe: dynamodb table check failed; ingest with dedupe on fails closed while it is retried", + "table", c.Table, "error", err) + } + a.add(component{name: "dedupe table check", run: func(ctx context.Context) error { + for wait := time.Second; ready() != nil; wait = min(2*wait, 30*time.Second) { + select { + case <-ctx.Done(): + return nil + case <-time.After(wait): + case <-retry: + } + if err := check(ctx); err != nil { + if ctx.Err() == nil { + slog.Error("dedupe: dynamodb table check failed again; ingest with dedupe on still fails closed", + "table", c.Table, "error", err) + } + continue + } + slog.Info("dedupe: dynamodb table check passed", "table", c.Table) + apply() + } + return nil + }}) + } + apply() + return nil +} + +// anyDedupeEnabled reports whether a served tenant has dedupe switched on. +func (a *App) anyDedupeEnabled() bool { + for _, store := range a.tenants.All() { + if store.DedupeEnabled() { + return true + } + } + return false +} diff --git a/internal/config/backends.go b/internal/config/backends.go index 239e95334..2c955e194 100644 --- a/internal/config/backends.go +++ b/internal/config/backends.go @@ -1,9 +1,11 @@ package config import ( + "errors" "fmt" "slices" "strings" + "time" ) // Each layer's implementation is chosen here, once, at boot: `.backend` @@ -68,20 +70,81 @@ func (c Cache) validate() error { // DedupeBackend names where ingest dedupe keeps the ids it has seen. type DedupeBackend string -// DedupePebble is the Pebble instance inside this process, under -// /pebble, opened while any tenant has dedupe on. -const DedupePebble DedupeBackend = "pebble" +const ( + // DedupePebble is the Pebble instance inside this process, under + // /pebble, opened while any tenant has dedupe on. Seen ids are + // per process. + DedupePebble DedupeBackend = "pebble" + // DedupeDynamoDB is one DynamoDB table every tenant and every process + // shares, configured by dedupe.dynamodb. + DedupeDynamoDB DedupeBackend = "dynamodb" +) -var dedupeBackends = []DedupeBackend{DedupePebble} +var dedupeBackends = []DedupeBackend{DedupePebble, DedupeDynamoDB} -// Dedupe selects the dedupe store. Whether a tenant dedupes, and on which -// field, are settings-directory keys, not this block's. +// Dedupe selects the dedupe store. Whether a tenant dedupes, on which field, +// and for how long are settings-directory keys, not this block's. type Dedupe struct { Backend DedupeBackend `yaml:"backend" env:"WH_DEDUPE_BACKEND"` + // Lease is how long a claimed id stays pending while its record is + // published; a claim its request never settles lapses after it. + Lease time.Duration `yaml:"lease" env:"WH_DEDUPE_LEASE"` + // ReserveConcurrency bounds the parallel calls one Reserve, Commit or + // Release makes to a remote backend, and sizes its idle connection pool + // to match. Pebble ignores it. + ReserveConcurrency int `yaml:"reserve_concurrency" env:"WH_DEDUPE_RESERVE_CONCURRENCY"` + DynamoDB DedupeDynamoDBConfig `yaml:"dynamodb"` +} + +// DedupeDynamoDBConfig is the dynamodb backend's block, read only when it is +// selected. Credentials are the AWS SDK's default chain (EKS Pod Identity, +// IRSA, AWS_* variables), never keys here. +type DedupeDynamoDBConfig struct { + // Table is the shared table; WaveHouse never creates it outside + // dynamodb-local. Required. + Table string `yaml:"table" env:"WH_DEDUPE_DYNAMODB_TABLE"` + // Region overrides the SDK chain's (AWS_REGION). + Region string `yaml:"region" env:"WH_DEDUPE_DYNAMODB_REGION"` + // Endpoint points the client at dynamodb-local. + Endpoint string `yaml:"endpoint" env:"WH_DEDUPE_DYNAMODB_ENDPOINT"` + Timeout time.Duration `yaml:"timeout" env:"WH_DEDUPE_DYNAMODB_TIMEOUT"` + MaxAttempts int `yaml:"max_attempts" env:"WH_DEDUPE_DYNAMODB_MAX_ATTEMPTS"` + RetryMode string `yaml:"retry_mode" env:"WH_DEDUPE_DYNAMODB_RETRY_MODE"` + // CreateTable creates the table at boot if it is missing. Development + // only: refused unless Endpoint is set. + CreateTable bool `yaml:"create_table" env:"WH_DEDUPE_DYNAMODB_CREATE_TABLE"` } func (d Dedupe) validate() error { - return checkBackend("dedupe.backend", "WH_DEDUPE_BACKEND", d.Backend, dedupeBackends) + if err := checkBackend("dedupe.backend", "WH_DEDUPE_BACKEND", d.Backend, dedupeBackends); err != nil { + return err + } + if d.Lease <= 0 { + return fmt.Errorf("dedupe.lease (WH_DEDUPE_LEASE) must be > 0, got %s", d.Lease) + } + if d.ReserveConcurrency <= 0 { + return fmt.Errorf("dedupe.reserve_concurrency (WH_DEDUPE_RESERVE_CONCURRENCY) must be > 0, got %d", d.ReserveConcurrency) + } + if d.Backend == DedupeDynamoDB { + return d.DynamoDB.validate() + } + return nil +} + +func (d DedupeDynamoDBConfig) validate() error { + switch { + case strings.TrimSpace(d.Table) == "": + return errors.New("dedupe.dynamodb.table (WH_DEDUPE_DYNAMODB_TABLE) is required when dedupe.backend is dynamodb") + case d.Timeout <= 0: + return fmt.Errorf("dedupe.dynamodb.timeout (WH_DEDUPE_DYNAMODB_TIMEOUT) must be > 0, got %s", d.Timeout) + case d.MaxAttempts <= 0: + return fmt.Errorf("dedupe.dynamodb.max_attempts (WH_DEDUPE_DYNAMODB_MAX_ATTEMPTS) must be > 0, got %d", d.MaxAttempts) + case d.RetryMode != "standard" && d.RetryMode != "adaptive": + return fmt.Errorf("dedupe.dynamodb.retry_mode (WH_DEDUPE_DYNAMODB_RETRY_MODE) %q: want standard or adaptive", d.RetryMode) + case d.CreateTable && d.Endpoint == "": + return errors.New("dedupe.dynamodb.create_table (WH_DEDUPE_DYNAMODB_CREATE_TABLE) is for dynamodb-local only: set dedupe.dynamodb.endpoint, or create the table with your infrastructure code") + } + return nil } // CoordBackend names where leases for singleton work (the sweeper) are held. @@ -116,13 +179,45 @@ func checkBackend[T ~string](key, env string, got T, valid []T) error { return fmt.Errorf("%s (%s) %q is not a backend this build has; valid: %s", key, env, got, strings.Join(names, ", ")) } -// validateBackends checks every layer's backend and its sub-block. +// embeddedDuplicateWindow is the embedded ingest stream's duplicate window, +// counted from the stored publish. It mirrors mq.EmbeddedDuplicateWindow, +// which config must not import; window_test.go pins the two. +const embeddedDuplicateWindow = 2 * time.Minute + +// maxEmbeddedLease is the longest dedupe.lease the duplicate window covers — +// the largest whole second satisfying the rule below. It is informational +// only: validateBackends checks the rule itself, not this constant, since +// the rule's ceiling steps at each whole second rather than moving linearly +// with the lease. +const maxEmbeddedLease = 59 * time.Second + +// ceilSecond rounds d up to the next whole second, as a DynamoDB claim's +// expiry does (epoch seconds, rounded up) — so a claim taken out just before +// the tick it is stamped with can stay live up to a second past the lease. +func ceilSecond(d time.Duration) time.Duration { + if r := d % time.Second; r != 0 { + d += time.Second - r + } + return d +} + +// validateBackends checks every layer's backend and its sub-block, then the +// rules that span two layers. func (c *Config) validateBackends() error { for _, check := range []func() error{c.MQ.validate, c.Cache.validate, c.Dedupe.validate, c.Coord.validate} { if err := check(); err != nil { return err } } + // A client obeying the in-flight 503's Retry-After (the whole lease) + // republishes at t0+lease at the earliest. But a claim can outlive its + // own lease by up to a second (DynamoDB rounds expiry up to the second), + // so the last such 503 can go out at t0+lease+1s, and the republish it + // asks for lands at t0+lease+1s+ceil(lease). That must still fall inside + // the embedded duplicate window: lease + ceil(lease) + 1s <= 2m. + if worst := c.Dedupe.Lease + ceilSecond(c.Dedupe.Lease) + time.Second; c.MQ.Backend == MQEmbedded && worst > embeddedDuplicateWindow { + return fmt.Errorf("dedupe.lease (WH_DEDUPE_LEASE) %s is over %s with the embedded mq: lease + ceil(lease) + 1s (%s) must fit its %s duplicate window, since a client obeying the in-flight 503's Retry-After can republish that late", c.Dedupe.Lease, maxEmbeddedLease, worst, embeddedDuplicateWindow) + } return nil } diff --git a/internal/config/backends_test.go b/internal/config/backends_test.go index 50d346d21..6b6081644 100644 --- a/internal/config/backends_test.go +++ b/internal/config/backends_test.go @@ -4,17 +4,18 @@ import ( "os" "path/filepath" "testing" + "time" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) // withDefaultBackends sets what defaults() would: a literal Config -// names no backend and no role, and Validate refuses that. +// names no backend, no role and no dedupe lease, and Validate refuses that. func withDefaultBackends(c Config) *Config { c.Roles = AllRoles() c.MQ.Backend, c.Cache.Backend = MQEmbedded, CacheLocal - c.Dedupe.Backend, c.Coord.Backend = DedupePebble, CoordLocal + c.Dedupe, c.Coord.Backend = defaults().Dedupe, CoordLocal return &c } @@ -29,6 +30,10 @@ func TestLoad_BackendDefaults(t *testing.T) { assert.Equal(t, MQEmbedded, cfg.MQ.Backend) assert.Equal(t, CacheLocal, cfg.Cache.Backend) assert.Equal(t, DedupePebble, cfg.Dedupe.Backend) + assert.Equal(t, Dedupe{ + Backend: DedupePebble, Lease: 30 * time.Second, ReserveConcurrency: 64, + DynamoDB: DedupeDynamoDBConfig{Timeout: 250 * time.Millisecond, MaxAttempts: 3, RetryMode: "standard"}, + }, cfg.Dedupe) assert.Equal(t, CoordLocal, cfg.Coord.Backend) assert.False(t, cfg.Distributed()) assert.True(t, cfg.NeedsDataDir()) @@ -113,7 +118,7 @@ func TestValidate_UnknownBackend(t *testing.T) { }{ {"mq", func(c *Config) { c.MQ.Backend = "kafka" }, `mq.backend (WH_MQ_BACKEND) "kafka" is not a backend this build has; valid: embedded`}, {"cache", func(c *Config) { c.Cache.Backend = "memcached" }, `cache.backend (WH_CACHE_BACKEND) "memcached" is not a backend this build has; valid: local, redis`}, - {"dedupe", func(c *Config) { c.Dedupe.Backend = "dynamodb" }, `dedupe.backend (WH_DEDUPE_BACKEND) "dynamodb" is not a backend this build has; valid: pebble`}, + {"dedupe", func(c *Config) { c.Dedupe.Backend = "redis" }, `dedupe.backend (WH_DEDUPE_BACKEND) "redis" is not a backend this build has; valid: pebble, dynamodb`}, {"coord", func(c *Config) { c.Coord.Backend = "nats" }, `coord.backend (WH_COORD_BACKEND) "nats" is not a backend this build has; valid: local`}, // The zero value, which a Config built without Load carries. {"empty", func(c *Config) { c.MQ.Backend = "" }, `mq.backend (WH_MQ_BACKEND) "" is not a backend`}, @@ -160,3 +165,143 @@ func TestNeedsDataDir(t *testing.T) { cfg.MQ.Backend = MQEmbedded assert.True(t, cfg.NeedsDataDir(), "the embedded mq keeps state under data_dir") } + +func TestLoad_DedupeDynamoDBFromEnv(t *testing.T) { + for k, v := range map[string]string{ + "WH_DEDUPE_BACKEND": "dynamodb", + "WH_DEDUPE_LEASE": "45s", + "WH_DEDUPE_RESERVE_CONCURRENCY": "16", + "WH_DEDUPE_DYNAMODB_TABLE": "wavehouse-dedupe-dev", + "WH_DEDUPE_DYNAMODB_REGION": "us-east-2", + "WH_DEDUPE_DYNAMODB_ENDPOINT": "http://localhost:8000", + "WH_DEDUPE_DYNAMODB_TIMEOUT": "1s", + "WH_DEDUPE_DYNAMODB_MAX_ATTEMPTS": "5", + "WH_DEDUPE_DYNAMODB_RETRY_MODE": "adaptive", + "WH_DEDUPE_DYNAMODB_CREATE_TABLE": "true", + } { + t.Setenv(k, v) + } + cfg, err := Load("nonexistent.yaml") + require.NoError(t, err) + assert.Equal(t, Dedupe{ + Backend: DedupeDynamoDB, Lease: 45 * time.Second, ReserveConcurrency: 16, + DynamoDB: DedupeDynamoDBConfig{ + Table: "wavehouse-dedupe-dev", Region: "us-east-2", Endpoint: "http://localhost:8000", + Timeout: time.Second, MaxAttempts: 5, RetryMode: "adaptive", CreateTable: true, + }, + }, cfg.Dedupe) + assert.True(t, cfg.NeedsDataDir(), "the embedded mq still keeps state under data_dir") +} + +func TestLoad_DedupeDynamoDBFromYAML(t *testing.T) { + t.Parallel() + path := filepath.Join(t.TempDir(), "config.yaml") + require.NoError(t, os.WriteFile(path, []byte(` +dedupe: + backend: dynamodb + lease: 20s + dynamodb: + table: wavehouse-dedupe-prod + timeout: 400ms +`), 0o600)) + cfg, err := Load(path) + require.NoError(t, err) + assert.Equal(t, DedupeDynamoDB, cfg.Dedupe.Backend) + assert.Equal(t, 20*time.Second, cfg.Dedupe.Lease) + assert.Equal(t, 64, cfg.Dedupe.ReserveConcurrency) + assert.Equal(t, DedupeDynamoDBConfig{ + Table: "wavehouse-dedupe-prod", Timeout: 400 * time.Millisecond, MaxAttempts: 3, RetryMode: "standard", + }, cfg.Dedupe.DynamoDB) +} + +func TestLoad_DedupeDynamoDBRefusesUnknownKeys(t *testing.T) { + t.Parallel() + path := filepath.Join(t.TempDir(), "config.yaml") + require.NoError(t, os.WriteFile(path, []byte(` +dedupe: + backend: dynamodb + dynamodb: + table: t + access_key_id: AKIA + redis: + addr: localhost:6379 +`), 0o600)) + _, err := Load(path) + require.Error(t, err) + assert.Contains(t, err.Error(), "dedupe.dynamodb.access_key_id, dedupe.redis") +} + +func TestUnboundEnv_KnowsTheDedupeVariables(t *testing.T) { + t.Parallel() + assert.Empty(t, unboundEnv([]string{ + "WH_DEDUPE_LEASE=30s", "WH_DEDUPE_RESERVE_CONCURRENCY=64", + "WH_DEDUPE_DYNAMODB_TABLE=t", "WH_DEDUPE_DYNAMODB_REGION=us-east-1", + "WH_DEDUPE_DYNAMODB_ENDPOINT=http://localhost:8000", "WH_DEDUPE_DYNAMODB_TIMEOUT=250ms", + "WH_DEDUPE_DYNAMODB_MAX_ATTEMPTS=3", "WH_DEDUPE_DYNAMODB_RETRY_MODE=standard", + "WH_DEDUPE_DYNAMODB_CREATE_TABLE=false", + })) +} + +func TestValidate_Dedupe(t *testing.T) { + t.Parallel() + dynamo := func(c *Config) { + c.Dedupe.Backend = DedupeDynamoDB + c.Dedupe.DynamoDB = DedupeDynamoDBConfig{Table: "t", Timeout: time.Second, MaxAttempts: 3, RetryMode: "standard"} + } + cases := []struct { + name string + set func(*Config) + want string // "" = valid + }{ + {"dynamodb", dynamo, ""}, + {"create_table with an endpoint", func(c *Config) { + dynamo(c) + c.Dedupe.DynamoDB.Endpoint, c.Dedupe.DynamoDB.CreateTable = "http://localhost:8000", true + }, ""}, + {"the block is not read under pebble", func(c *Config) { c.Dedupe.DynamoDB = DedupeDynamoDBConfig{CreateTable: true} }, ""}, + {"lease at the cap", func(c *Config) { c.Dedupe.Lease = 59 * time.Second }, ""}, + {"lease just past the cap", func(c *Config) { c.Dedupe.Lease = 59*time.Second + 100*time.Millisecond }, "is over 59s with the embedded mq"}, + {"create_table without an endpoint", func(c *Config) { + dynamo(c) + c.Dedupe.DynamoDB.CreateTable = true + }, "dedupe.dynamodb.create_table (WH_DEDUPE_DYNAMODB_CREATE_TABLE) is for dynamodb-local only"}, + {"no table", func(c *Config) { dynamo(c); c.Dedupe.DynamoDB.Table = " " }, "dedupe.dynamodb.table (WH_DEDUPE_DYNAMODB_TABLE) is required"}, + {"retry mode", func(c *Config) { dynamo(c); c.Dedupe.DynamoDB.RetryMode = "legacy" }, `retry_mode (WH_DEDUPE_DYNAMODB_RETRY_MODE) "legacy"`}, + {"zero timeout", func(c *Config) { dynamo(c); c.Dedupe.DynamoDB.Timeout = 0 }, "dedupe.dynamodb.timeout (WH_DEDUPE_DYNAMODB_TIMEOUT) must be > 0"}, + {"negative timeout", func(c *Config) { dynamo(c); c.Dedupe.DynamoDB.Timeout = -time.Second }, "dedupe.dynamodb.timeout"}, + {"zero attempts", func(c *Config) { dynamo(c); c.Dedupe.DynamoDB.MaxAttempts = 0 }, "dedupe.dynamodb.max_attempts (WH_DEDUPE_DYNAMODB_MAX_ATTEMPTS) must be > 0"}, + {"negative attempts", func(c *Config) { dynamo(c); c.Dedupe.DynamoDB.MaxAttempts = -1 }, "dedupe.dynamodb.max_attempts"}, + {"no retry mode", func(c *Config) { dynamo(c); c.Dedupe.DynamoDB.RetryMode = "" }, `retry_mode (WH_DEDUPE_DYNAMODB_RETRY_MODE) ""`}, + {"zero lease", func(c *Config) { c.Dedupe.Lease = 0 }, "dedupe.lease (WH_DEDUPE_LEASE) must be > 0, got 0s"}, + {"negative lease", func(c *Config) { c.Dedupe.Lease = -time.Second }, "dedupe.lease (WH_DEDUPE_LEASE) must be > 0"}, + {"zero concurrency", func(c *Config) { c.Dedupe.ReserveConcurrency = 0 }, "dedupe.reserve_concurrency (WH_DEDUPE_RESERVE_CONCURRENCY) must be > 0"}, + {"negative concurrency", func(c *Config) { c.Dedupe.ReserveConcurrency = -1 }, "dedupe.reserve_concurrency"}, + {"lease of a minute", func(c *Config) { c.Dedupe.Lease = time.Minute }, "dedupe.lease (WH_DEDUPE_LEASE) 1m0s is over 59s with the embedded mq: lease + ceil(lease) + 1s (2m1s) must fit its 2m0s duplicate window"}, + {"lease at the old 59.5s cap", func(c *Config) { c.Dedupe.Lease = 59*time.Second + 500*time.Millisecond }, "is over 59s with the embedded mq"}, + {"lease at the duplicate window", func(c *Config) { c.Dedupe.Lease = 2 * time.Minute }, "is over 59s with the embedded mq"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + cfg := defaultBackends() + tc.set(&cfg) + err := cfg.Validate() + if tc.want == "" { + require.NoError(t, err) + return + } + require.Error(t, err) + assert.Contains(t, err.Error(), tc.want) + }) + } +} + +func TestNeedsDataDir_DynamoDBDedupe(t *testing.T) { + t.Parallel() + cfg := defaultBackends() + cfg.Dedupe.Backend = DedupeDynamoDB + assert.True(t, cfg.NeedsDataDir(), "the embedded mq keeps state under data_dir") + cfg.MQ.Backend = "shared" + assert.False(t, cfg.NeedsDataDir(), "neither a shared mq nor dynamodb dedupe keeps state under data_dir") + assert.Len(t, cfg.Warnings(), 1, "only the local cache warning: dynamodb dedupe is shared") +} diff --git a/internal/config/config.go b/internal/config/config.go index 0bc70f2ab..c6abd263d 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -264,8 +264,11 @@ func defaults() Config { MaxValueBytes: 1 << 20, CompressMinBytes: 1 << 10, VersionTTL: 168 * time.Hour, }, }, - Dedupe: Dedupe{Backend: DedupePebble}, - Coord: Coord{Backend: CoordLocal}, + Dedupe: Dedupe{ + Backend: DedupePebble, Lease: 30 * time.Second, ReserveConcurrency: 64, + DynamoDB: DedupeDynamoDBConfig{Timeout: 250 * time.Millisecond, MaxAttempts: 3, RetryMode: "standard"}, + }, + Coord: Coord{Backend: CoordLocal}, OTel: OTel{ Traces: OTelTraces{Enabled: true, SampleRate: 1.0}, Metrics: OTelMetrics{Enabled: true}, diff --git a/internal/config/defaults_test.go b/internal/config/defaults_test.go index 1a8fba337..ddcba538e 100644 --- a/internal/config/defaults_test.go +++ b/internal/config/defaults_test.go @@ -51,35 +51,53 @@ var zeroCases = []zeroCase{ // refusedZeros are the non-zero defaults whose zero Validate refuses: written // in the file, the zero must reach Validate rather than become the default. +// also holds the keys a sub-block's zero needs to be read at all. var refusedZeros = []struct { key string zero any err string + also map[string]any }{ - {"server.port", 0, "server.port 0 out of range"}, - {"mq.backend", "", `mq.backend (WH_MQ_BACKEND) ""`}, - {"cache.backend", "", `cache.backend (WH_CACHE_BACKEND) ""`}, - {"dedupe.backend", "", `dedupe.backend (WH_DEDUPE_BACKEND) ""`}, - {"coord.backend", "", `coord.backend (WH_COORD_BACKEND) ""`}, - {"roles", []string{}, "roles (WH_ROLES) is empty"}, + {"server.port", 0, "server.port 0 out of range", nil}, + {"mq.backend", "", `mq.backend (WH_MQ_BACKEND) ""`, nil}, + {"cache.backend", "", `cache.backend (WH_CACHE_BACKEND) ""`, nil}, + {"dedupe.backend", "", `dedupe.backend (WH_DEDUPE_BACKEND) ""`, nil}, + {"coord.backend", "", `coord.backend (WH_COORD_BACKEND) ""`, nil}, + {"roles", []string{}, "roles (WH_ROLES) is empty", nil}, + {"dedupe.lease", "0s", "dedupe.lease (WH_DEDUPE_LEASE) must be > 0", nil}, + {"dedupe.reserve_concurrency", 0, "dedupe.reserve_concurrency (WH_DEDUPE_RESERVE_CONCURRENCY) must be > 0", nil}, + {"dedupe.dynamodb.timeout", "0s", "dedupe.dynamodb.timeout (WH_DEDUPE_DYNAMODB_TIMEOUT) must be > 0", dynamoSelected}, + {"dedupe.dynamodb.max_attempts", 0, "dedupe.dynamodb.max_attempts (WH_DEDUPE_DYNAMODB_MAX_ATTEMPTS) must be > 0", dynamoSelected}, + {"dedupe.dynamodb.retry_mode", "", `dedupe.dynamodb.retry_mode (WH_DEDUPE_DYNAMODB_RETRY_MODE) ""`, dynamoSelected}, } -// yamlAt renders a file setting key to value, plus otel.enabled: true so -// the test can tell the file was read. -func yamlAt(t *testing.T, key string, value any) string { +// dynamoSelected is what the dedupe.dynamodb block needs to be read. +var dynamoSelected = map[string]any{"dedupe.backend": "dynamodb", "dedupe.dynamodb.table": "t"} + +// yamlAt renders a file setting key to value, and each dotted key of also to +// its value, plus otel.enabled: true so the test can tell the file was read. +func yamlAt(t *testing.T, key string, value any, also ...map[string]any) string { t.Helper() tree := map[string]any{"otel": map[string]any{"enabled": true}} - node := tree - parts := strings.Split(key, ".") - for _, p := range parts[:len(parts)-1] { - sub, ok := node[p].(map[string]any) - if !ok { - sub = map[string]any{} - node[p] = sub + set := func(key string, value any) { + node := tree + parts := strings.Split(key, ".") + for _, p := range parts[:len(parts)-1] { + sub, ok := node[p].(map[string]any) + if !ok { + sub = map[string]any{} + node[p] = sub + } + node = sub + } + node[parts[len(parts)-1]] = value + } + for _, m := range also { + for k, v := range m { + set(k, v) } - node = sub } - node[parts[len(parts)-1]] = value + set(key, value) out, err := yaml.Marshal(tree) require.NoError(t, err) return string(out) @@ -146,7 +164,7 @@ func TestLoad_YAMLZeroIsRefused(t *testing.T) { for _, tc := range refusedZeros { t.Run(tc.key, func(t *testing.T) { t.Parallel() - _, err := Load(writeYAML(t, yamlAt(t, tc.key, tc.zero))) + _, err := Load(writeYAML(t, yamlAt(t, tc.key, tc.zero, tc.also))) require.ErrorContains(t, err, tc.err, "the zero reaches Validate instead of becoming the default") }) } diff --git a/internal/config/window_test.go b/internal/config/window_test.go new file mode 100644 index 000000000..9d2559cd2 --- /dev/null +++ b/internal/config/window_test.go @@ -0,0 +1,15 @@ +package config + +import ( + "testing" + + "github.com/Wave-RF/WaveHouse/internal/mq" + "github.com/stretchr/testify/assert" +) + +// config must not import internal/mq (it would pull NATS into every +// importer of config), so the lease cap mirrors the window; this pins them. +func TestEmbeddedDuplicateWindow_MatchesMQ(t *testing.T) { + t.Parallel() + assert.Equal(t, mq.EmbeddedDuplicateWindow, embeddedDuplicateWindow) +} diff --git a/internal/dedupe/conformance_test.go b/internal/dedupe/conformance_test.go new file mode 100644 index 000000000..afadae283 --- /dev/null +++ b/internal/dedupe/conformance_test.go @@ -0,0 +1,54 @@ +package dedupe_test + +import ( + "sync" + "testing" + "time" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/dedupe/dedupetest" +) + +// fakeClock is a clock tests move by hand. +type fakeClock struct { + mu sync.Mutex + now time.Time +} + +func (c *fakeClock) Now() time.Time { + c.mu.Lock() + defer c.mu.Unlock() + return c.now +} + +func (c *fakeClock) Advance(d time.Duration) { + c.mu.Lock() + defer c.mu.Unlock() + c.now = c.now.Add(d) +} + +func TestEmbedded_Conformance(t *testing.T) { + t.Parallel() + dedupetest.Run(t, func(t *testing.T) dedupetest.Harness { + e := dedupe.NewEmbedded(t.TempDir()) + clock := &fakeClock{now: time.Now()} + dedupe.SetClock(e, clock.Now) + return dedupetest.Harness{ + Factory: e.Tenant, + Advance: clock.Advance, + FailNextReserve: func(n int) { dedupe.FailNextReserve(e, n) }, + } + }) +} + +// The suite's sleeping path, which a backend without an injectable clock +// takes, on the real clock. +func TestEmbedded_ConformanceRealClock(t *testing.T) { + t.Parallel() + if testing.Short() { + t.Skip("sleeps past leases") + } + dedupetest.Run(t, func(t *testing.T) dedupetest.Harness { + return dedupetest.Harness{Factory: dedupe.NewEmbedded(t.TempDir()).Tenant} + }) +} diff --git a/internal/dedupe/dedupe.go b/internal/dedupe/dedupe.go index c9fdb7f46..68061ef4a 100644 --- a/internal/dedupe/dedupe.go +++ b/internal/dedupe/dedupe.go @@ -1,13 +1,89 @@ package dedupe -import "context" +import ( + "context" + "time" +) -// Deduplicator checks whether an event has been seen before and marks it. -type Deduplicator interface { - // CheckAndMark returns true if the event was already seen (duplicate). - // If not seen, it atomically marks the event as seen. - CheckAndMark(ctx context.Context, eventID string) (isDuplicate bool, err error) +// DefaultLease is how long a Claimed key stays pending when the caller names +// no lease: long enough to cover a publish, short enough that a request that +// died mid-publish does not hold the id for long. +const DefaultLease = 30 * time.Second + +// Key is one record's dedupe identity inside a tenant's store. The tenant is +// bound by the store (Stores.For), so a Key never carries it. +type Key struct { + Table string + ID string +} + +// Status is Reserve's verdict for one key. +type Status uint8 + +const ( + // Claimed is a first sighting within retention. The caller now holds a + // pending claim and must Commit it once the record is published, or + // Release it if the publish definitely failed. An abandoned claim lapses + // after the lease. + Claimed Status = iota + 1 + // Duplicate means the key was committed earlier and has not expired: skip + // the record. Managed also answers it for a key repeated inside one + // Reserve call, after its first occurrence, whatever the first answered. + Duplicate + // InFlight means another request holds a live claim on the key. Its + // outcome is not known yet, so the caller answers 503 and the client + // retries. + InFlight +) - // Close releases resources held by the deduplicator. +func (s Status) String() string { + switch s { + case Claimed: + return "claimed" + case Duplicate: + return "duplicate" + case InFlight: + return "in_flight" + default: + return "unknown" + } +} + +// Claim is Reserve's answer for one key. Token is the backend's proof of +// ownership, opaque to callers; Release compares it. +type Claim struct { + Key Key + Status Status + Token string +} + +// Deduplicator is a tenant's store of seen ids. Callers reach every backend +// through Managed, which hands a backend distinct keys, a lease > 0, and +// only Claimed claims to Commit and Release — a backend may assume all +// three, and Managed's callers get the behaviour below either way. +// +// Reserve is atomic per key: of any number of concurrent Reserves for the +// same key — in this process or any other sharing the backend — at most one +// returns Claimed. It returns one Claim per key, in input order. On error it +// has released every claim it knows it made; a write whose outcome the +// error left unknown (a timeout, a cancelled call) may still land +// afterwards, and then holds its key InFlight until the lease ends, like an +// abandoned claim. The error wraps ErrUnavailable when retrying later can +// succeed (throttled, timed out, backend unreachable). +// +// Commit makes Claimed claims duplicates for retention (0 = no expiry) and +// ignores claims of any other status. It is unconditional: a commit that +// lands after its lease lapsed and another request re-claimed the key is +// still correct, because the committing request did publish. +// +// Release gives up the Claimed claims it still owns (token match); a claim +// that has lapsed or been re-claimed is left alone. +// +// There is deliberately no read-only check: every caller that asks "have I +// seen this" needs the claim too, and a separate read is how #390 happened. +type Deduplicator interface { + Reserve(ctx context.Context, keys []Key, lease time.Duration) ([]Claim, error) + Commit(ctx context.Context, claims []Claim, retention time.Duration) error + Release(ctx context.Context, claims []Claim) error Close() error } diff --git a/internal/dedupe/dedupe_test.go b/internal/dedupe/dedupe_test.go new file mode 100644 index 000000000..a2655bb3c --- /dev/null +++ b/internal/dedupe/dedupe_test.go @@ -0,0 +1,19 @@ +package dedupe + +import ( + "testing" + + "github.com/stretchr/testify/assert" +) + +func TestStatus_String(t *testing.T) { + t.Parallel() + for s, want := range map[Status]string{ + Claimed: "claimed", + Duplicate: "duplicate", + InFlight: "in_flight", + Status(0): "unknown", + } { + assert.Equal(t, want, s.String()) + } +} diff --git a/internal/dedupe/dedupetest/dedupetest.go b/internal/dedupe/dedupetest/dedupetest.go new file mode 100644 index 000000000..94e0b27c6 --- /dev/null +++ b/internal/dedupe/dedupetest/dedupetest.go @@ -0,0 +1,423 @@ +// Package dedupetest is the conformance suite every dedupe backend runs: the +// Deduplicator contract (dedupe.go) as tests, driven through the production +// path — a backend's Factory and the Managed switch it returns — so a backend +// that passes here behaves the same under ingest as every other. +package dedupetest + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "fmt" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// Harness is one fresh backend under test. +type Harness struct { + // Factory builds a tenant's store over the backend. The suite switches + // each store it builds on and closes it at cleanup. + Factory dedupe.Factory + // Peer, if set, builds a tenant's store over the same data through a + // second client — another process's view, for backends that have one. + // nil uses Factory. + Peer dedupe.Factory + // Advance moves the backend's clock forward by d. nil makes the suite + // sleep instead, which is why its leases and retentions are whole + // seconds: a backend may store expiry at one-second resolution. + Advance func(d time.Duration) + // FailNextReserve, if set, makes the backend's next Reserve fail after it + // has claimed n keys. nil skips the case that needs it. + FailNextReserve func(n int) +} + +// Run runs every case, each against a backend newHarness builds fresh. +func Run(t *testing.T, newHarness func(t *testing.T) Harness) { + t.Helper() + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + t.Parallel() + c.run(t, &suite{Harness: newHarness(t)}) + }) + } +} + +// Mark is the old check-and-mark in one call, for tests that only need an id +// seen: it reserves k and commits it with no expiry, reporting whether k was +// already committed. A key another request holds is an error. +func Mark(ctx context.Context, d dedupe.Deduplicator, k dedupe.Key) (duplicate bool, err error) { + claims, err := d.Reserve(ctx, []dedupe.Key{k}, dedupe.DefaultLease) + if err != nil { + return false, err + } + switch claims[0].Status { + case dedupe.Duplicate: + return true, nil + case dedupe.Claimed: + return false, d.Commit(ctx, claims, 0) + case dedupe.InFlight: + } + return false, fmt.Errorf("key %v is %s", k, claims[0].Status) +} + +const ( + lease = time.Second + // long outlives every case, so only a deliberate pass lapses it. + long = time.Hour +) + +type suite struct { + Harness +} + +func (s *suite) open(t *testing.T, build dedupe.Factory, id tenant.ID) dedupe.Deduplicator { + t.Helper() + m := build(id) + require.NoError(t, m.Apply(true)) + t.Cleanup(func() { _ = m.Close() }) + return m +} + +// store opens tenant id's store; peer opens it through the second client. +func (s *suite) store(t *testing.T, id tenant.ID) dedupe.Deduplicator { + t.Helper() + return s.open(t, s.Factory, id) +} + +func (s *suite) peer(t *testing.T, id tenant.ID) dedupe.Deduplicator { + t.Helper() + if s.Peer == nil { + return s.store(t, id) + } + return s.open(t, s.Peer, id) +} + +// pass lets d go by, plus a second's margin for a backend that stores expiry +// in whole seconds. +func (s *suite) pass(d time.Duration) { + d += time.Second + if s.Advance != nil { + s.Advance(d) + return + } + time.Sleep(d) +} + +func reserve(t *testing.T, d dedupe.Deduplicator, lease time.Duration, keys ...dedupe.Key) []dedupe.Claim { + t.Helper() + claims, err := d.Reserve(t.Context(), keys, lease) + require.NoError(t, err) + require.Len(t, claims, len(keys)) + for i, c := range claims { + require.Equal(t, keys[i], c.Key, "claim %d answers its own key, in input order", i) + } + return claims +} + +func statuses(claims []dedupe.Claim) []dedupe.Status { + out := make([]dedupe.Status, len(claims)) + for i, c := range claims { + out[i] = c.Status + } + return out +} + +func key(id string) dedupe.Key { return dedupe.Key{Table: "events", ID: id} } + +var cases = []struct { + name string + run func(t *testing.T, s *suite) +}{ + {"claim then commit is a duplicate", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + c := reserve(t, d, long, key("e1")) + require.Equal(t, dedupe.Claimed, c[0].Status) + assert.NotEmpty(t, c[0].Token) + require.NoError(t, d.Commit(t.Context(), c, 0)) + assert.Equal(t, dedupe.Duplicate, reserve(t, d, long, key("e1"))[0].Status) + assert.Equal(t, dedupe.Claimed, reserve(t, d, long, key("e2"))[0].Status, "distinct ids are independent") + }}, + {"a released claim can be claimed again", func(t *testing.T, s *suite) { + // #384: a publish that failed releases the id, and the client's retry + // goes through. + d := s.store(t, "acme") + c := reserve(t, d, long, key("e1")) + require.NoError(t, d.Release(t.Context(), c)) + c = reserve(t, d, long, key("e1")) + assert.Equal(t, dedupe.Claimed, c[0].Status) + require.NoError(t, d.Commit(t.Context(), c, 0)) + assert.Equal(t, dedupe.Duplicate, reserve(t, d, long, key("e1"))[0].Status) + }}, + {"a live claim is in flight to everyone else", func(t *testing.T, s *suite) { + d, p := s.store(t, "acme"), s.peer(t, "acme") + c := reserve(t, d, long, key("e1")) + assert.Equal(t, dedupe.InFlight, reserve(t, d, long, key("e1"))[0].Status) + assert.Equal(t, dedupe.InFlight, reserve(t, p, long, key("e1"))[0].Status, "and to another client") + require.NoError(t, d.Commit(t.Context(), c, 0)) + assert.Equal(t, dedupe.Duplicate, reserve(t, p, long, key("e1"))[0].Status, "the peer sees the commit") + }}, + {"an abandoned claim lapses after its lease", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + reserve(t, d, lease, key("e1")) + s.pass(lease) + assert.Equal(t, dedupe.Claimed, reserve(t, d, long, key("e1"))[0].Status) + }}, + {"a commit expires after its retention", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + c := reserve(t, d, long, key("brief"), key("kept")) + require.NoError(t, d.Commit(t.Context(), c[:1], time.Second)) + require.NoError(t, d.Commit(t.Context(), c[1:], 0)) + assert.Equal(t, []dedupe.Status{dedupe.Duplicate, dedupe.Duplicate}, statuses(reserve(t, d, long, key("brief"), key("kept")))) + s.pass(time.Second) + assert.Equal(t, []dedupe.Status{dedupe.Claimed, dedupe.Duplicate}, statuses(reserve(t, d, long, key("brief"), key("kept"))), + "retention 0 never expires") + }}, + {"concurrent reserves of one key claim it once", func(t *testing.T, s *suite) { + // #390: two requests carrying one id must not both publish, checked + // across many fresh keys since a narrow lock window can miss one. + const n = 16 + const keys = 20 + d, p := s.store(t, "acme"), s.peer(t, "acme") + race := func(k dedupe.Key) []dedupe.Claim { + out := make([]dedupe.Claim, n) + var wg sync.WaitGroup + for i := range n { + store := d + if i%2 == 1 { + store = p + } + wg.Go(func() { + c, err := store.Reserve(context.Background(), []dedupe.Key{k}, long) + if assert.NoError(t, err) { + out[i] = c[0] + } + }) + } + wg.Wait() + return out + } + for round := range keys { + k := key(fmt.Sprintf("race-%d", round)) + var winner []dedupe.Claim + for _, c := range race(k) { + if c.Status == dedupe.Claimed { + winner = append(winner, c) + } else { + assert.Equal(t, dedupe.InFlight, c.Status, "round %d", round) + } + } + require.Len(t, winner, 1, "round %d: exactly one reserve claims the key", round) + require.NoError(t, d.Commit(t.Context(), winner, 0)) + for _, c := range race(k) { + assert.Equal(t, dedupe.Duplicate, c.Status, "round %d", round) + } + } + }}, + {"a reserve racing a commit never claims, and settles to duplicate once it lands", func(t *testing.T, s *suite) { + // Commit must land durably before it drops the pending claim; a + // racing Reserve must never see Claimed, only Duplicate once it + // returns. A start barrier holds Commit until every worker has + // made its first call, so low GOMAXPROCS can't starve them out of + // overlapping it at all. + const workers = 4 + const rounds = 8 + d, p := s.store(t, "acme"), s.peer(t, "acme") + for round := range rounds { + k := key(fmt.Sprintf("commit-race-%d", round)) + c := reserve(t, d, long, k) + var stop atomic.Bool + var claimed atomic.Int64 + var ready sync.WaitGroup + ready.Add(workers) + var wg sync.WaitGroup + for i := range workers { + store := d + if i%2 == 1 { + store = p + } + wg.Go(func() { + first := true + for !stop.Load() { + got, err := store.Reserve(context.Background(), []dedupe.Key{k}, long) + if first { + first = false + ready.Done() + } + if !assert.NoError(t, err) { + return + } + switch got[0].Status { + case dedupe.Claimed: + claimed.Add(1) + case dedupe.InFlight, dedupe.Duplicate: + default: + t.Errorf("round %d: unexpected status %v", round, got[0].Status) + } + } + }) + } + ready.Wait() + require.NoError(t, d.Commit(t.Context(), c, 0)) + stop.Store(true) + wg.Wait() + assert.Zero(t, claimed.Load(), "round %d: a reserve claimed a key mid-commit", round) + for i := range workers { + store := d + if i%2 == 1 { + store = p + } + got, err := store.Reserve(context.Background(), []dedupe.Key{k}, long) + if assert.NoError(t, err) { + assert.Equal(t, dedupe.Duplicate, got[0].Status, "round %d: reserve %d after commit", round, i) + } + } + } + }}, + {"a key repeated in one call is claimed once", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + c := reserve(t, d, long, key("a"), key("b"), key("a"), key("a")) + assert.Equal(t, []dedupe.Status{dedupe.Claimed, dedupe.Claimed, dedupe.Duplicate, dedupe.Duplicate}, statuses(c)) + require.NoError(t, d.Commit(t.Context(), c, 0), "commit ignores the repeats") + assert.Equal(t, []dedupe.Status{dedupe.Duplicate, dedupe.Duplicate}, statuses(reserve(t, d, long, key("a"), key("b")))) + }}, + {"answers keep input order in a large call", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + keys := make([]dedupe.Key, 300) + for i := range keys { + keys[i] = key(fmt.Sprint(i)) + } + var odd []dedupe.Key + for i := 1; i < len(keys); i += 2 { + odd = append(odd, keys[i]) + } + require.NoError(t, d.Commit(t.Context(), reserve(t, d, long, odd...), 0)) + for i, c := range reserve(t, d, long, keys...) { + want := dedupe.Claimed + if i%2 == 1 { + want = dedupe.Duplicate + } + assert.Equal(t, want, c.Status, "key %d", i) + } + }}, + {"tables and tenants have their own keyspace", func(t *testing.T, s *suite) { + acme, globex := s.store(t, "acme"), s.store(t, "globex") + // "ab"+"c" and "a"+"bc" would be one key were table and id just + // joined; tenants "a"/"ab" likewise. + first := []dedupe.Key{{Table: "clicks", ID: "e1"}, {Table: "ab", ID: "c"}} + require.NoError(t, acme.Commit(t.Context(), reserve(t, acme, long, first...), 0)) + assert.Equal(t, []dedupe.Status{dedupe.Claimed, dedupe.Claimed}, + statuses(reserve(t, acme, long, dedupe.Key{Table: "views", ID: "e1"}, dedupe.Key{Table: "a", ID: "bc"})), "#222: another table's id") + assert.Equal(t, []dedupe.Status{dedupe.Claimed, dedupe.Claimed}, + statuses(reserve(t, globex, long, first...)), "another tenant's ids") + a, ab := s.store(t, "a"), s.store(t, "ab") + require.NoError(t, a.Commit(t.Context(), reserve(t, a, long, dedupe.Key{Table: "bt", ID: "e1"}), 0)) + assert.Equal(t, dedupe.Claimed, reserve(t, ab, long, dedupe.Key{Table: "t", ID: "e1"})[0].Status) + assert.Equal(t, []dedupe.Status{dedupe.Duplicate, dedupe.Duplicate}, + statuses(reserve(t, acme, long, first...)), "and still duplicates in their own") + }}, + {"long ids and ids that look hashed stay distinct", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + base := strings.Repeat("x", 2*dedupe.MaxIDBytes) + longA, longB := key(base+"a"), key(base+"b") + // hashLike spells longA's own stored hashed form, escaped: it fits + // verbatim and stays distinct only if the '#' the hashed form writes + // raw never gets escaped like an ordinary field. + sum := sha256.Sum256([]byte(base + "a")) + hashLike := key("#" + hex.EncodeToString(sum[:])) + first := reserve(t, d, long, longA, hashLike) + assert.Equal(t, []dedupe.Status{dedupe.Claimed, dedupe.Claimed}, statuses(first), + "both fresh — a collision would answer the second InFlight") + require.NoError(t, d.Commit(t.Context(), first, 0)) + assert.Equal(t, []dedupe.Status{dedupe.Duplicate, dedupe.Claimed, dedupe.Duplicate}, + statuses(reserve(t, d, long, longA, longB, hashLike))) + }}, + {"a late commit after a re-claim still lands", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + first := reserve(t, d, lease, key("e1")) + s.pass(lease) + second := reserve(t, d, long, key("e1")) + require.Equal(t, dedupe.Claimed, second[0].Status) + require.NoError(t, d.Commit(t.Context(), first, 0), "the first request did publish") + require.NoError(t, d.Release(t.Context(), second), "the second gives up; the commit stands") + assert.Equal(t, dedupe.Duplicate, reserve(t, d, long, key("e1"))[0].Status) + }}, + {"a stale release leaves the new claimant alone", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + first := reserve(t, d, lease, key("e1")) + s.pass(lease) + second := reserve(t, d, long, key("e1")) + require.NoError(t, d.Release(t.Context(), first)) + assert.Equal(t, dedupe.InFlight, reserve(t, d, long, key("e1"))[0].Status, "the second claim is still live") + require.NoError(t, d.Commit(t.Context(), second, 0)) + assert.Equal(t, dedupe.Duplicate, reserve(t, d, long, key("e1"))[0].Status) + }}, + {"a failed reserve leaves nothing claimed", func(t *testing.T, s *suite) { + if s.FailNextReserve == nil { + t.Skip("the backend has no failure hook") + } + d := s.store(t, "acme") + s.FailNextReserve(1) + _, err := d.Reserve(t.Context(), []dedupe.Key{key("a"), key("b"), key("c")}, long) + require.Error(t, err) + assert.Equal(t, []dedupe.Status{dedupe.Claimed, dedupe.Claimed, dedupe.Claimed}, + statuses(reserve(t, d, long, key("a"), key("b"), key("c")))) + }}, + {"empty calls are no-ops", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + c, err := d.Reserve(t.Context(), nil, long) + require.NoError(t, err) + assert.Empty(t, c) + require.NoError(t, d.Commit(t.Context(), nil, 0)) + require.NoError(t, d.Release(t.Context(), nil)) + dup := []dedupe.Claim{{Key: key("e1"), Status: dedupe.Duplicate}, {Key: key("e2"), Status: dedupe.InFlight}} + require.NoError(t, d.Commit(t.Context(), dup, 0), "only Claimed claims commit") + assert.Equal(t, dedupe.Claimed, reserve(t, d, long, key("e1"))[0].Status) + }}, + {"any table name is its own keyspace", func(t *testing.T, s *suite) { + d := s.store(t, "acme") + // seen[i] and fresh[i] differ only in where table ends and id + // begins, or in a byte an escaped key could confuse with its + // escape: each pair would share one key under a layout that + // separated the fields without escaping them. + seen := []dedupe.Key{ + {Table: "a", ID: "b\x00c"}, + {Table: "a\x00", ID: "b"}, + {Table: "", ID: "\x01a"}, + {Table: "\xff\xfe", ID: "e1"}, + {Table: "tab\tle \n", ID: "e1"}, + {Table: "a/b", ID: "c"}, + {Table: "a/b", ID: "d"}, + {Table: "t", ID: "%23x"}, + } + fresh := []dedupe.Key{ + {Table: "a\x00b", ID: "c"}, + {Table: "a", ID: "\x00b"}, + {Table: "\x01", ID: "a"}, + {Table: "\xff", ID: "\xfee1"}, + {Table: "tab\tle", ID: " \ne1"}, + {Table: "a%2Fb", ID: "c"}, + {Table: "a", ID: "b/d"}, + {Table: "t", ID: "#x"}, + } + first := reserve(t, d, long, seen...) + for _, c := range first { + require.Equal(t, dedupe.Claimed, c.Status, "%q shares a key with another seen key", c.Key) + } + require.NoError(t, d.Commit(t.Context(), first, 0)) + for _, c := range reserve(t, d, long, fresh...) { + assert.Equal(t, dedupe.Claimed, c.Status, "%q", c.Key) + } + for _, c := range reserve(t, d, long, seen...) { + assert.Equal(t, dedupe.Duplicate, c.Status, "%q", c.Key) + } + }}, +} diff --git a/internal/dedupe/dynamodb.go b/internal/dedupe/dynamodb.go new file mode 100644 index 000000000..e6cb55d2a --- /dev/null +++ b/internal/dedupe/dynamodb.go @@ -0,0 +1,710 @@ +package dedupe + +import ( + "context" + "crypto/rand" + "errors" + "fmt" + "log/slog" + mathrand "math/rand/v2" + "net/http" + "strconv" + "sync" + "time" + + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/aws/retry" + awshttp "github.com/aws/aws-sdk-go-v2/aws/transport/http" + "github.com/aws/aws-sdk-go-v2/config" + "github.com/aws/aws-sdk-go-v2/service/dynamodb" + "github.com/aws/aws-sdk-go-v2/service/dynamodb/types" + "github.com/aws/smithy-go" + "go.opentelemetry.io/otel" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/metric" + "golang.org/x/sync/errgroup" + + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// The table's attributes. pk is the key from key.go and the only key +// attribute; ex is the table's TTL attribute. +const ( + attrKey = "pk" + attrState = "st" + attrExpiry = "ex" + attrToken = "tk" + + statePending = "1" + stateCommitted = "2" + + // A claim is live while now < ex; one whose ex has passed is absent + // to Reserve, whether or not TTL has deleted it yet. + condReserve = "attribute_not_exists(pk) OR ex <= :now" + condRelease = "tk = :tk AND st = :pending" + + // batchWriteMax is BatchWriteItem's per-call item limit. + batchWriteMax = 25 + // commitRounds bounds the BatchWriteItem rounds one chunk gets before + // its still-unprocessed (or still-throttled) items fail the Commit. + commitRounds = 8 + // commitBase and commitCeiling bound the jittered wait between Commit + // rounds; retryBase is the one the SDK retryer's backoff doubles from. + commitBase = 25 * time.Millisecond + commitCeiling = 200 * time.Millisecond + retryBase = 25 * time.Millisecond + tokenBytes = 16 + + // opReserve is the operation the breaker watches: Release and Commit + // answers say nothing about whether a new Reserve would get through. + opReserve = "put_item" +) + +// DynamoConfig is the DynamoDB backend's wiring. Credentials are never here: +// the SDK's default chain finds them (EKS Pod Identity or IRSA in a pod, the +// environment or a profile locally). +type DynamoConfig struct { + // Table is the shared table every tenant's keys live in. Required. + Table string + // Region overrides the SDK chain's region (AWS_REGION) when set. + Region string + // Endpoint points the client at dynamodb-local. Tests and development + // only; it is also what unlocks CreateTable. + Endpoint string + // Timeout bounds each DynamoDB call, its SDK retries included. + // 0 = 250ms. The retries' jittered backoff is capped so that together + // it waits at most half of Timeout (retryBackoff). + Timeout time.Duration + // MaxAttempts is the SDK retryer's attempts per call. 0 = 3. + MaxAttempts int + // RetryMode is "standard" (default) or "adaptive", which also rate-limits + // the client after throttles. + RetryMode string + // ReserveConcurrency bounds the parallel calls one Reserve, Commit or + // Release makes, and sizes the client's idle connection pool to match + // (never below the SDK's default of 10 per host). + // 0 = 64. + ReserveConcurrency int +} + +func (c DynamoConfig) withDefaults() DynamoConfig { + if c.Timeout <= 0 { + c.Timeout = 250 * time.Millisecond + } + if c.MaxAttempts <= 0 { + c.MaxAttempts = 3 + } + if c.RetryMode == "" { + c.RetryMode = "standard" + } + if c.ReserveConcurrency <= 0 { + c.ReserveConcurrency = 64 + } + return c +} + +// dynamoAPI is the part of *dynamodb.Client the backend calls, so a unit test +// can inject throttles and unprocessed items. +type dynamoAPI interface { + PutItem(context.Context, *dynamodb.PutItemInput, ...func(*dynamodb.Options)) (*dynamodb.PutItemOutput, error) + BatchWriteItem(context.Context, *dynamodb.BatchWriteItemInput, ...func(*dynamodb.Options)) (*dynamodb.BatchWriteItemOutput, error) + DeleteItem(context.Context, *dynamodb.DeleteItemInput, ...func(*dynamodb.Options)) (*dynamodb.DeleteItemOutput, error) + DescribeTable(context.Context, *dynamodb.DescribeTableInput, ...func(*dynamodb.Options)) (*dynamodb.DescribeTableOutput, error) + DescribeTimeToLive(context.Context, *dynamodb.DescribeTimeToLiveInput, ...func(*dynamodb.Options)) (*dynamodb.DescribeTimeToLiveOutput, error) + CreateTable(context.Context, *dynamodb.CreateTableInput, ...func(*dynamodb.Options)) (*dynamodb.CreateTableOutput, error) + UpdateTimeToLive(context.Context, *dynamodb.UpdateTimeToLiveInput, ...func(*dynamodb.Options)) (*dynamodb.UpdateTimeToLiveOutput, error) +} + +// Dynamo is the DynamoDB implementation: every tenant's keys in one shared +// table, so pods sharing the table share seen ids and Reserve's conditional +// write is atomic across all of them. WaveHouse never creates the table in +// production; CreateTable is for dynamodb-local. +type Dynamo struct { + api dynamoAPI + cfg DynamoConfig + now func() time.Time + breaker *breaker + metrics dynamoMetrics + // commitBackoff is the wait before retrying the attempt'th round of + // unprocessed or throttled items. + commitBackoff func(attempt int) time.Duration +} + +// NewDynamo builds the backend over a client from the SDK's default config +// chain. extra is appended to the chain's options (a test's static +// credentials or HTTP client, say). It dials nothing: Check does. +func NewDynamo(ctx context.Context, cfg DynamoConfig, extra ...func(*config.LoadOptions) error) (*Dynamo, error) { + if cfg.Table == "" { + return nil, errors.New("dedupe: dynamodb table is required") + } + cfg = cfg.withDefaults() + retryer, err := newRetryer(cfg) + if err != nil { + return nil, err + } + opts := []func(*config.LoadOptions) error{config.WithRetryer(retryer), config.WithHTTPClient(newHTTPClient(cfg))} + if cfg.Region != "" { + opts = append(opts, config.WithRegion(cfg.Region)) + } + awsCfg, err := config.LoadDefaultConfig(ctx, append(opts, extra...)...) + if err != nil { + return nil, fmt.Errorf("dedupe: aws config: %w", err) + } + if awsCfg.Region == "" { + return nil, errors.New("dedupe: dynamodb region is not set: set dedupe.dynamodb.region or AWS_REGION") + } + client := dynamodb.NewFromConfig(awsCfg, func(o *dynamodb.Options) { + if cfg.Endpoint != "" { + o.BaseEndpoint = aws.String(cfg.Endpoint) + } + }) + return newDynamo(client, cfg), nil +} + +// newHTTPClient keeps an idle connection for every call one Reserve can have +// in flight, and never fewer than the SDK's defaults: with its 10 per host, a +// wide Reserve would dial most of its puts afresh. +func newHTTPClient(cfg DynamoConfig) *awshttp.BuildableClient { + return awshttp.NewBuildableClient().WithTransportOptions(func(tr *http.Transport) { + tr.MaxIdleConnsPerHost = max(tr.MaxIdleConnsPerHost, cfg.ReserveConcurrency) + tr.MaxIdleConns = max(tr.MaxIdleConns, cfg.ReserveConcurrency) + }) +} + +func newRetryer(cfg DynamoConfig) (func() aws.Retryer, error) { + standard := func(o *retry.StandardOptions) { + o.MaxAttempts = cfg.MaxAttempts + o.Backoff = retryBackoff(cfg) + } + switch cfg.RetryMode { + case "standard": + return func() aws.Retryer { return retry.NewStandard(standard) }, nil + case "adaptive": + return func() aws.Retryer { + return retry.NewAdaptiveMode(func(o *retry.AdaptiveModeOptions) { + o.StandardOptions = append(o.StandardOptions, standard) + }) + }, nil + } + return nil, fmt.Errorf("dedupe: dynamodb retry_mode %q: want standard or adaptive", cfg.RetryMode) +} + +// retryBackoff is the SDK retryer's wait before a retry: full jitter, so +// puts throttled together do not retry in lockstep, under a ceiling that +// doubles from retryBase up to Timeout/(2·(MaxAttempts-1)). A call's retries +// then wait at most half its Timeout in all, so a throttled call ends on its +// last attempt's answer (ErrUnavailable, the throttle as its cause) unless +// the attempts themselves take the other half. +func retryBackoff(cfg DynamoConfig) retry.BackoffDelayerFunc { + ceiling := cfg.Timeout / time.Duration(2*max(cfg.MaxAttempts-1, 1)) + return func(attempt int, _ error) (time.Duration, error) { + return fullJitter(retryBase, ceiling, attempt), nil + } +} + +// fullJitter is uniform over [0, min(base·2^attempt, ceiling)]. +func fullJitter(base, ceiling time.Duration, attempt int) time.Duration { + d := min(base< 0 { + ex = &types.AttributeValueMemberN{Value: strconv.FormatInt(expiresAt(s.d.now(), retention), 10)} + } + // BatchWriteItem refuses a key twice in one call; a caller merging + // claims from two Reserves could hand one over twice. + seen := make(map[string]bool, len(claims)) + writes := make([]types.WriteRequest, 0, len(claims)) + for _, c := range claims { + pk := string(AppendKey(nil, s.prefix, c.Key)) + if seen[pk] { + continue + } + seen[pk] = true + item := map[string]types.AttributeValue{ + attrKey: &types.AttributeValueMemberS{Value: pk}, + attrState: &types.AttributeValueMemberN{Value: stateCommitted}, + attrToken: &types.AttributeValueMemberB{Value: []byte(c.Token)}, + } + if ex != nil { + item[attrExpiry] = ex + } + writes = append(writes, types.WriteRequest{PutRequest: &types.PutRequest{Item: item}}) + } + // Every chunk is attempted whatever another's fate: these records are + // already published, and an uncommitted id lets a retry publish again. + chunks := (len(writes) + batchWriteMax - 1) / batchWriteMax + return forEach(chunks, s.d.cfg.ReserveConcurrency, func(i int) error { + return s.commitChunk(ctx, writes[i*batchWriteMax:min((i+1)*batchWriteMax, len(writes))]) + }) +} + +func (s *dynamoStore) commitChunk(ctx context.Context, writes []types.WriteRequest) error { + for attempt := 0; ; attempt++ { + var unprocessed []types.WriteRequest + err := s.d.call(ctx, "batch_write_item", func(ctx context.Context) error { + out, err := s.d.api.BatchWriteItem(ctx, &dynamodb.BatchWriteItemInput{ + RequestItems: map[string][]types.WriteRequest{s.d.cfg.Table: writes}, + }) + if err == nil { + unprocessed = out.UnprocessedItems[s.d.cfg.Table] + } + return err + }) + switch { + case err == nil: + case errors.Is(err, ErrUnavailable): + // DynamoDB throttles a batch whole only when it processed none of + // it, and a timeout leaves its fate unknown: retry it whole, as a + // round that left every item unprocessed. The puts are idempotent. + unprocessed = writes + default: + return err + } + if len(unprocessed) == 0 { + return nil + } + if attempt+1 >= commitRounds { + if err != nil { + return err + } + return fmt.Errorf("%w: dynamodb batch_write_item: %d items still unprocessed", ErrUnavailable, len(unprocessed)) + } + if err == nil { + s.d.metrics.unprocessed.Add(ctx, int64(len(unprocessed))) + } + writes = unprocessed + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(s.d.commitBackoff(attempt)): + } + } +} + +// Release deletes each claim's item only while it is still that claim's +// pending item; a failed condition means the key lapsed, was re-claimed or +// was committed, and is left alone. +func (s *dynamoStore) Release(ctx context.Context, claims []Claim) error { + if len(claims) == 0 { + return nil + } + // Every claim is attempted: one left behind holds its id for a lease. + return forEach(len(claims), s.d.cfg.ReserveConcurrency, func(i int) error { + c := claims[i] + err := s.d.call(ctx, "delete_item", func(ctx context.Context) error { + _, err := s.d.api.DeleteItem(ctx, &dynamodb.DeleteItemInput{ + TableName: &s.d.cfg.Table, + Key: map[string]types.AttributeValue{attrKey: &types.AttributeValueMemberS{Value: string(AppendKey(nil, s.prefix, c.Key))}}, + ConditionExpression: aws.String(condRelease), + ExpressionAttributeValues: map[string]types.AttributeValue{ + ":tk": &types.AttributeValueMemberB{Value: []byte(c.Token)}, + ":pending": &types.AttributeValueMemberN{Value: statePending}, + }, + }) + return err + }) + var gone *types.ConditionalCheckFailedException + if errors.As(err, &gone) { + return nil + } + return err + }) +} + +// Close is a no-op: the client is the Dynamo's, shared by every tenant. +func (s *dynamoStore) Close() error { return nil } + +// forEach runs do for every index, at most limit at once, and joins the +// errors: one failure never stops the rest. +func forEach(n, limit int, do func(i int) error) error { + errs := make([]error, n) + var g errgroup.Group + g.SetLimit(limit) + for i := range n { + g.Go(func() error { errs[i] = do(i); return nil }) + } + _ = g.Wait() + return errors.Join(errs...) +} + +// expiresAt is t+d in epoch seconds rounded up, so a claim or commit never +// ends before it was asked to: TTL attributes are whole seconds. +func expiresAt(t time.Time, d time.Duration) int64 { + end := t.Add(d) + sec := end.Unix() + if end.Nanosecond() > 0 { + sec++ + } + return sec +} + +func newToken() string { + b := make([]byte, tokenBytes) + _, _ = rand.Read(b) // crypto/rand.Read never fails + return string(b) +} + +// classify maps a DynamoDB error onto the contract: a condition failure is +// returned as is for the caller to read, anything retrying later can cure +// wraps ErrUnavailable, and the rest — a missing table, denied access, a +// malformed request — is a configuration bug. Ingest answers ErrUnavailable +// with a retryable 503 and a configuration bug with a 500. +func classify(op string, err error) error { + if err == nil { + return nil + } + var cond *types.ConditionalCheckFailedException + if errors.As(err, &cond) { + return err + } + if transient(err) { + return fmt.Errorf("%w: dynamodb %s: %w", ErrUnavailable, op, err) + } + return fmt.Errorf("dynamodb %s: %w", op, err) +} + +func transient(err error) bool { + if errors.Is(err, context.DeadlineExceeded) { + return true + } + if (retry.RetryableConnectionError{}).IsErrorRetryable(err) == aws.TrueTernary { + return true + } + var api smithy.APIError + if !errors.As(err, &api) { + return false + } + code := api.ErrorCode() + if _, ok := retry.DefaultThrottleErrorCodes[code]; ok { + return true + } + if _, ok := retry.DefaultRetryableErrorCodes[code]; ok { + return true + } + switch code { + case "InternalServerError", "ServiceUnavailable", "ReplicatedWriteConflictException": + return true + } + return api.ErrorFault() == smithy.FaultServer +} + +// breaker short-circuits Reserve for a second after breakerTrips consecutive +// unavailable answers inside a second, so a throttled or unreachable table +// fails requests fast instead of spending every one's full timeout. +type breaker struct { + mu sync.Mutex + now func() time.Time + fails int + since time.Time + openUntil time.Time +} + +const ( + breakerTrips = 5 + breakerWindow = time.Second + breakerCool = time.Second +) + +var errBreakerOpen = fmt.Errorf("%w: dynamodb is failing; short-circuited", ErrUnavailable) + +func newBreaker(now func() time.Time) *breaker { return &breaker{now: now} } + +func (b *breaker) allow() error { + b.mu.Lock() + defer b.mu.Unlock() + if b.now().Before(b.openUntil) { + return errBreakerOpen + } + return nil +} + +func (b *breaker) record(err error) { + b.mu.Lock() + defer b.mu.Unlock() + if !errors.Is(err, ErrUnavailable) { + b.fails = 0 + return + } + now := b.now() + if b.fails == 0 || now.Sub(b.since) > breakerWindow { + b.fails, b.since = 0, now + } + b.fails++ + if b.fails >= breakerTrips { + b.fails = 0 + b.openUntil = now.Add(breakerCool) + } +} + +type dynamoMetrics struct { + requests metric.Int64Counter + duration metric.Float64Histogram + unprocessed metric.Int64Counter + shorted metric.Int64Counter +} + +func newDynamoMetrics() dynamoMetrics { + meter := otel.Meter("wavehouse-dedupe") + requests, _ := meter.Int64Counter("wavehouse_dedupe_dynamodb_requests_total", + metric.WithDescription("DynamoDB dedupe requests by operation and outcome (ok, condition_failed, unavailable, canceled, error)")) + duration, _ := meter.Float64Histogram("wavehouse_dedupe_dynamodb_request_duration_seconds", + metric.WithDescription("DynamoDB dedupe request latency, SDK retries included"), metric.WithUnit("s")) + unprocessed, _ := meter.Int64Counter("wavehouse_dedupe_dynamodb_unprocessed_items_total", + metric.WithDescription("Commit items DynamoDB left unprocessed and the backend retried")) + shorted, _ := meter.Int64Counter("wavehouse_dedupe_dynamodb_short_circuits_total", + metric.WithDescription("Reserves refused without a request while DynamoDB was failing")) + return dynamoMetrics{requests: requests, duration: duration, unprocessed: unprocessed, shorted: shorted} +} + +func (m dynamoMetrics) record(ctx context.Context, op string, took time.Duration, err error) { + outcome := "ok" + var cond *types.ConditionalCheckFailedException + switch { + case err == nil: + case errors.As(err, &cond): + outcome = "condition_failed" + case errors.Is(err, context.Canceled): + outcome = "canceled" + case errors.Is(err, ErrUnavailable): + outcome = "unavailable" + default: + outcome = "error" + } + ctx = context.WithoutCancel(ctx) + m.requests.Add(ctx, 1, metric.WithAttributes(attribute.String("op", op), attribute.String("outcome", outcome))) + m.duration.Record(ctx, took.Seconds(), metric.WithAttributes(attribute.String("op", op))) +} + +func (m dynamoMetrics) shortCircuit(ctx context.Context) { m.shorted.Add(ctx, 1) } diff --git a/internal/dedupe/dynamodb_bench_test.go b/internal/dedupe/dynamodb_bench_test.go new file mode 100644 index 000000000..ad7239edf --- /dev/null +++ b/internal/dedupe/dynamodb_bench_test.go @@ -0,0 +1,77 @@ +//go:build dynamobench + +// Manual latency benchmark for the DynamoDB backend, never run by CI. Point it +// at an existing table (the credentials and region come from the SDK chain): +// +// DEDUPE_BENCH_TABLE=wavehouse-dedupe-dev go test -tags dynamobench \ +// -run '^$' -bench Dynamo -benchtime 2000x ./internal/dedupe/ +// +// DEDUPE_BENCH_ENDPOINT=http://localhost:8000 runs it against dynamodb-local +// instead, creating the table there. +package dedupe + +import ( + "fmt" + "os" + "sync/atomic" + "testing" + "time" +) + +var benchSeq atomic.Uint64 + +func benchDynamo(b *testing.B) *Managed { + b.Helper() + cfg := DynamoConfig{Table: os.Getenv("DEDUPE_BENCH_TABLE"), Endpoint: os.Getenv("DEDUPE_BENCH_ENDPOINT")} + if cfg.Table == "" { + b.Skip("DEDUPE_BENCH_TABLE is not set") + } + if cfg.Endpoint != "" { + cfg.Timeout = 5 * time.Second // dynamodb-local is far slower than the service + } + d, err := NewDynamo(b.Context(), cfg) + if err != nil { + b.Fatal(err) + } + if cfg.Endpoint != "" { + if err := d.CreateTable(b.Context()); err != nil { + b.Fatal(err) + } + } + if err := d.Check(b.Context()); err != nil { + b.Fatal(err) + } + m := d.Tenant("bench") + if err := m.Apply(true); err != nil { + b.Fatal(err) + } + return m +} + +// benchKeys are n ids no run has used, with a short retention so the table +// forgets them. +func benchKeys(n int) []Key { + run := time.Now().UnixNano() + out := make([]Key, n) + for i := range out { + out[i] = Key{Table: "bench", ID: fmt.Sprintf("%d-%d", run, benchSeq.Add(1))} + } + return out +} + +func benchReserveCommit(b *testing.B, window int) { + m := benchDynamo(b) + b.ResetTimer() + for b.Loop() { + claims, err := m.Reserve(b.Context(), benchKeys(window), DefaultLease) + if err != nil { + b.Fatal(err) + } + if err := m.Commit(b.Context(), claims, time.Hour); err != nil { + b.Fatal(err) + } + } +} + +func BenchmarkDynamo_ReserveCommit1(b *testing.B) { benchReserveCommit(b, 1) } +func BenchmarkDynamo_ReserveCommit256(b *testing.B) { benchReserveCommit(b, 256) } diff --git a/internal/dedupe/dynamodb_test.go b/internal/dedupe/dynamodb_test.go new file mode 100644 index 000000000..a61132590 --- /dev/null +++ b/internal/dedupe/dynamodb_test.go @@ -0,0 +1,773 @@ +package dedupe + +import ( + "context" + "errors" + "fmt" + "io" + "maps" + "net" + "net/http" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/aws/retry" + awshttp "github.com/aws/aws-sdk-go-v2/aws/transport/http" + "github.com/aws/aws-sdk-go-v2/config" + "github.com/aws/aws-sdk-go-v2/credentials" + "github.com/aws/aws-sdk-go-v2/service/dynamodb" + "github.com/aws/aws-sdk-go-v2/service/dynamodb/types" + "github.com/aws/smithy-go" + smithyhttp "github.com/aws/smithy-go/transport/http" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// fakeDynamo answers each operation through its func, or with success when +// that is nil. The DynamoDB semantics themselves are tested against +// dynamodb-local (tests/integration); this is for the error paths it cannot +// produce. +type fakeDynamo struct { + put func(context.Context, *dynamodb.PutItemInput) (*dynamodb.PutItemOutput, error) + batch func(context.Context, *dynamodb.BatchWriteItemInput) (*dynamodb.BatchWriteItemOutput, error) + del func(context.Context, *dynamodb.DeleteItemInput) (*dynamodb.DeleteItemOutput, error) + describe func() (*dynamodb.DescribeTableOutput, error) + ttl func() (*dynamodb.DescribeTimeToLiveOutput, error) +} + +func (f *fakeDynamo) PutItem(ctx context.Context, in *dynamodb.PutItemInput, _ ...func(*dynamodb.Options)) (*dynamodb.PutItemOutput, error) { + if f.put == nil { + return &dynamodb.PutItemOutput{}, nil + } + return f.put(ctx, in) +} + +func (f *fakeDynamo) BatchWriteItem(ctx context.Context, in *dynamodb.BatchWriteItemInput, _ ...func(*dynamodb.Options)) (*dynamodb.BatchWriteItemOutput, error) { + if err := ctx.Err(); err != nil { + return nil, err + } + if f.batch == nil { + return &dynamodb.BatchWriteItemOutput{}, nil + } + return f.batch(ctx, in) +} + +func (f *fakeDynamo) DeleteItem(ctx context.Context, in *dynamodb.DeleteItemInput, _ ...func(*dynamodb.Options)) (*dynamodb.DeleteItemOutput, error) { + if err := ctx.Err(); err != nil { + return nil, err + } + if f.del == nil { + return &dynamodb.DeleteItemOutput{}, nil + } + return f.del(ctx, in) +} + +func (f *fakeDynamo) DescribeTable(context.Context, *dynamodb.DescribeTableInput, ...func(*dynamodb.Options)) (*dynamodb.DescribeTableOutput, error) { + return f.describe() +} + +func (f *fakeDynamo) DescribeTimeToLive(context.Context, *dynamodb.DescribeTimeToLiveInput, ...func(*dynamodb.Options)) (*dynamodb.DescribeTimeToLiveOutput, error) { + return f.ttl() +} + +func (f *fakeDynamo) CreateTable(context.Context, *dynamodb.CreateTableInput, ...func(*dynamodb.Options)) (*dynamodb.CreateTableOutput, error) { + return nil, errors.New("not used") +} + +func (f *fakeDynamo) UpdateTimeToLive(context.Context, *dynamodb.UpdateTimeToLiveInput, ...func(*dynamodb.Options)) (*dynamodb.UpdateTimeToLiveOutput, error) { + return nil, errors.New("not used") +} + +func apiErr(code string, fault smithy.ErrorFault) error { + return &smithy.GenericAPIError{Code: code, Message: "injected", Fault: fault} +} + +func openFake(t *testing.T, f *fakeDynamo) (*Dynamo, Deduplicator) { + t.Helper() + return openFakeWith(t, f, DynamoConfig{Table: "dedupe"}) +} + +func openFakeWith(t *testing.T, f *fakeDynamo, cfg DynamoConfig) (*Dynamo, Deduplicator) { + t.Helper() + d := newDynamo(f, cfg) + d.commitBackoff = func(int) time.Duration { return 0 } + m := d.Tenant("acme") + require.NoError(t, m.Apply(true)) + t.Cleanup(func() { _ = m.Close() }) + return d, m +} + +func keys(ids ...string) []Key { + out := make([]Key, len(ids)) + for i, id := range ids { + out[i] = Key{Table: "events", ID: id} + } + return out +} + +func TestClassify(t *testing.T) { + t.Parallel() + for _, tc := range []struct { + name string + err error + unavailable bool + }{ + {"throttled", apiErr("ThrottlingException", smithy.FaultClient), true}, + {"over provisioned throughput", &types.ProvisionedThroughputExceededException{}, true}, + {"account request limit", apiErr("RequestLimitExceeded", smithy.FaultClient), true}, + {"internal error", &types.InternalServerError{}, true}, + {"unknown server fault", apiErr("Whatever", smithy.FaultServer), true}, + {"request timeout", apiErr("RequestTimeoutException", smithy.FaultClient), true}, + {"multi-region write conflict", &types.ReplicatedWriteConflictException{}, true}, + {"deadline", fmt.Errorf("op: %w", context.DeadlineExceeded), true}, + {"connection refused", &smithyhttp.RequestSendError{Err: &net.OpError{Op: "dial", Err: errors.New("refused")}}, true}, + {"missing table", &types.ResourceNotFoundException{}, false}, + {"access denied", apiErr("AccessDeniedException", smithy.FaultClient), false}, + {"validation", apiErr("ValidationException", smithy.FaultClient), false}, + {"caller went away", context.Canceled, false}, + } { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + err := classify("put_item", tc.err) + assert.ErrorIs(t, err, tc.err, "the cause stays reachable") + assert.Equal(t, tc.unavailable, errors.Is(err, ErrUnavailable)) + }) + } + assert.NoError(t, classify("put_item", nil)) + ccf := &types.ConditionalCheckFailedException{} + assert.Same(t, error(ccf), classify("put_item", ccf), "a condition failure is an answer, not an error") +} + +func TestDynamo_ReserveReadsTheHeldItem(t *testing.T) { + t.Parallel() + _, m := openFake(t, &fakeDynamo{put: func(_ context.Context, in *dynamodb.PutItemInput) (*dynamodb.PutItemOutput, error) { + id := in.Item[attrKey].(*types.AttributeValueMemberS).Value + switch id[len(id)-1] { + case 'd': + return nil, &types.ConditionalCheckFailedException{Item: map[string]types.AttributeValue{attrState: &types.AttributeValueMemberN{Value: stateCommitted}}} + case 'f': + return nil, &types.ConditionalCheckFailedException{Item: map[string]types.AttributeValue{attrState: &types.AttributeValueMemberN{Value: statePending}}} + } + assert.Equal(t, condReserve, aws.ToString(in.ConditionExpression)) + assert.Equal(t, types.ReturnValuesOnConditionCheckFailureAllOld, in.ReturnValuesOnConditionCheckFailure) + return &dynamodb.PutItemOutput{}, nil + }}) + claims, err := m.Reserve(t.Context(), keys("new", "old", "inf"), time.Minute) + require.NoError(t, err) + assert.Equal(t, []Status{Claimed, Duplicate, InFlight}, []Status{claims[0].Status, claims[1].Status, claims[2].Status}) + assert.Len(t, claims[0].Token, tokenBytes) + assert.Empty(t, claims[1].Token) +} + +// An SDK retry of a put whose first attempt was applied fails its condition +// on the put's own item: that is the caller's claim, not another request's. +func TestDynamo_RetriedPutKeepsItsOwnClaim(t *testing.T) { + t.Parallel() + var mu sync.Mutex + putTokens := map[string]string{} + var released []string + fake := &fakeDynamo{ + put: func(_ context.Context, in *dynamodb.PutItemInput) (*dynamodb.PutItemOutput, error) { + id := idOf(in.Item[attrKey]) + mu.Lock() + putTokens[id] = string(in.Item[attrToken].(*types.AttributeValueMemberB).Value) + mu.Unlock() + switch id { + case "k0": + return nil, &types.ConditionalCheckFailedException{Item: in.Item} + case "k1": + theirs := maps.Clone(in.Item) + theirs[attrToken] = &types.AttributeValueMemberB{Value: []byte("theirs")} + return nil, &types.ConditionalCheckFailedException{Item: theirs} + } + return nil, &types.InternalServerError{} + }, + del: func(_ context.Context, in *dynamodb.DeleteItemInput) (*dynamodb.DeleteItemOutput, error) { + id := idOf(in.Key[attrKey]) + mu.Lock() + defer mu.Unlock() + assert.Equal(t, putTokens[id], string(in.ExpressionAttributeValues[":tk"].(*types.AttributeValueMemberB).Value)) + released = append(released, id) + return &dynamodb.DeleteItemOutput{}, nil + }, + } + _, m := openFakeWith(t, fake, DynamoConfig{Table: "dedupe", ReserveConcurrency: 1}) + + claims, err := m.Reserve(t.Context(), keys("k0", "k1"), time.Minute) + require.NoError(t, err) + assert.Equal(t, []Status{Claimed, InFlight}, []Status{claims[0].Status, claims[1].Status}, "a pending item is InFlight only under another token") + assert.Equal(t, putTokens["k0"], claims[0].Token) + + // k0 is sent and answered before k2 fails, so the undo owns it. + _, err = m.Reserve(t.Context(), keys("k0", "k2"), time.Minute) + require.ErrorIs(t, err, ErrUnavailable) + mu.Lock() + defer mu.Unlock() + assert.Contains(t, released, "k0", "the failed Reserve's undo releases the retried put's own item") +} + +func TestDynamo_FailedReserveReleasesEveryPutThatMayHaveLanded(t *testing.T) { + t.Parallel() + var mu sync.Mutex + putTokens := map[string]string{} + var released []string + _, m := openFake(t, &fakeDynamo{ + put: func(_ context.Context, in *dynamodb.PutItemInput) (*dynamodb.PutItemOutput, error) { + id := in.Item[attrKey].(*types.AttributeValueMemberS).Value + mu.Lock() + putTokens[id] = string(in.Item[attrToken].(*types.AttributeValueMemberB).Value) + mu.Unlock() + switch id[len(id)-3:] { + case "dup": + return nil, &types.ConditionalCheckFailedException{Item: map[string]types.AttributeValue{attrState: &types.AttributeValueMemberN{Value: stateCommitted}}} + case "bad": + return nil, &types.InternalServerError{} + } + return &dynamodb.PutItemOutput{}, nil + }, + del: func(_ context.Context, in *dynamodb.DeleteItemInput) (*dynamodb.DeleteItemOutput, error) { + id := in.Key[attrKey].(*types.AttributeValueMemberS).Value + mu.Lock() + defer mu.Unlock() + assert.Equal(t, putTokens[id], string(in.ExpressionAttributeValues[":tk"].(*types.AttributeValueMemberB).Value), "released by the token it was put with") + released = append(released, id[len(id)-3:]) + return &dynamodb.DeleteItemOutput{}, nil + }, + }) + _, err := m.Reserve(t.Context(), keys("ok1", "dup", "bad", "ok2"), time.Minute) + require.ErrorIs(t, err, ErrUnavailable) + var sent []string + for id := range putTokens { + if id[len(id)-3:] != "dup" { + sent = append(sent, id[len(id)-3:]) + } + } + assert.ElementsMatch(t, sent, released, "every sent put but the duplicate, which was never ours") + assert.Contains(t, released, "bad", "the failed put may have landed") +} + +func TestDynamo_CommitRetriesUnprocessedItems(t *testing.T) { + t.Parallel() + var calls atomic.Int64 + var mu sync.Mutex + written := map[string]int{} + heldBack := map[string]bool{} + _, m := openFake(t, &fakeDynamo{batch: func(_ context.Context, in *dynamodb.BatchWriteItemInput) (*dynamodb.BatchWriteItemOutput, error) { + calls.Add(1) + reqs := in.RequestItems["dedupe"] + assert.LessOrEqual(t, len(reqs), batchWriteMax) + // Leave the last item of every call unprocessed once. + mu.Lock() + defer mu.Unlock() + var left []types.WriteRequest + for i, r := range reqs { + pk := r.PutRequest.Item[attrKey].(*types.AttributeValueMemberS).Value + assert.Equal(t, stateCommitted, r.PutRequest.Item[attrState].(*types.AttributeValueMemberN).Value) + assert.Contains(t, r.PutRequest.Item, attrExpiry) + if i == len(reqs)-1 && !heldBack[pk] && len(reqs) > 1 { + heldBack[pk] = true + left = append(left, r) + continue + } + written[pk]++ + } + return &dynamodb.BatchWriteItemOutput{UnprocessedItems: map[string][]types.WriteRequest{"dedupe": left}}, nil + }}) + ids := make([]string, 60) + for i := range ids { + ids[i] = fmt.Sprint(i) + } + claims := make([]Claim, 0, len(ids)+1) + for _, k := range keys(ids...) { + claims = append(claims, Claim{Key: k, Status: Claimed, Token: "t"}) + } + claims = append(claims, claims[0]) + require.NoError(t, m.Commit(t.Context(), claims, time.Hour)) + assert.Len(t, written, 60, "a key handed over twice is written once") + for pk, n := range written { + assert.Equal(t, 1, n, "%q", pk) + } + assert.Equal(t, int64(6), calls.Load(), "3 chunks, each retried once") +} + +func TestDynamo_CommitGivesUpOnItemsThatStayUnprocessed(t *testing.T) { + t.Parallel() + _, m := openFake(t, &fakeDynamo{batch: func(_ context.Context, in *dynamodb.BatchWriteItemInput) (*dynamodb.BatchWriteItemOutput, error) { + return &dynamodb.BatchWriteItemOutput{UnprocessedItems: in.RequestItems}, nil + }}) + err := m.Commit(t.Context(), []Claim{{Key: keys("a")[0], Status: Claimed, Token: "t"}}, 0) + require.ErrorIs(t, err, ErrUnavailable) +} + +// DynamoDB answers a BatchWriteItem it processed none of with a throttle, +// not with every item unprocessed: the chunk gets the same rounds either way. +func TestDynamo_CommitRetriesAFailedBatch(t *testing.T) { + t.Parallel() + throttle := &types.ProvisionedThroughputExceededException{} + for _, tc := range []struct { + name string + fails int + err error + calls int64 + committed bool + unavailable bool + }{ + {"throttled, then through", 3, throttle, 4, true, false}, + {"throttled every round", 1 << 10, throttle, commitRounds, false, true}, + {"timed out, then through", 1, fmt.Errorf("op: %w", context.DeadlineExceeded), 2, true, false}, + {"a configuration bug is final", 1 << 10, &types.ResourceNotFoundException{}, 1, false, false}, + } { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + var calls atomic.Int64 + _, m := openFake(t, &fakeDynamo{batch: func(_ context.Context, in *dynamodb.BatchWriteItemInput) (*dynamodb.BatchWriteItemOutput, error) { + assert.Len(t, in.RequestItems["dedupe"], 2, "the whole batch, every round") + if calls.Add(1) <= int64(tc.fails) { + return nil, tc.err + } + return &dynamodb.BatchWriteItemOutput{}, nil + }}) + var claims []Claim + for _, k := range keys("a", "b") { + claims = append(claims, Claim{Key: k, Status: Claimed, Token: "t"}) + } + err := m.Commit(t.Context(), claims, 0) + assert.Equal(t, tc.calls, calls.Load()) + if tc.committed { + require.NoError(t, err) + return + } + require.ErrorIs(t, err, tc.err, "the last round's cause is kept") + assert.Equal(t, tc.unavailable, errors.Is(err, ErrUnavailable)) + }) + } +} + +func TestDynamo_ReleaseTreatsAFailedConditionAsDone(t *testing.T) { + t.Parallel() + _, m := openFake(t, &fakeDynamo{del: func(_ context.Context, in *dynamodb.DeleteItemInput) (*dynamodb.DeleteItemOutput, error) { + assert.Equal(t, condRelease, aws.ToString(in.ConditionExpression)) + id := in.Key[attrKey].(*types.AttributeValueMemberS).Value + if id[len(id)-1] == 'x' { + return nil, &types.ResourceNotFoundException{} + } + return nil, &types.ConditionalCheckFailedException{} + }}) + claim := func(id string) []Claim { return []Claim{{Key: keys(id)[0], Status: Claimed, Token: "t"}} } + require.NoError(t, m.Release(t.Context(), claim("gone"))) + err := m.Release(t.Context(), claim("x")) + require.Error(t, err) + assert.False(t, errors.Is(err, ErrUnavailable)) +} + +func TestDynamo_BreakerShortCircuitsReserve(t *testing.T) { + t.Parallel() + var puts atomic.Int64 + var down atomic.Bool + down.Store(true) + d, m := openFake(t, &fakeDynamo{put: func(context.Context, *dynamodb.PutItemInput) (*dynamodb.PutItemOutput, error) { + puts.Add(1) + if down.Load() { + return nil, &types.ProvisionedThroughputExceededException{} + } + return &dynamodb.PutItemOutput{}, nil + }}) + now := time.Unix(1_000_000, 0) + var clock sync.Mutex + d.breaker.now = func() time.Time { clock.Lock(); defer clock.Unlock(); return now } + for range breakerTrips { + _, err := m.Reserve(t.Context(), keys("a"), time.Minute) + require.ErrorIs(t, err, ErrUnavailable) + } + _, err := m.Reserve(t.Context(), keys("a"), time.Minute) + require.ErrorIs(t, err, errBreakerOpen) + assert.Equal(t, int64(breakerTrips), puts.Load(), "the open breaker sent nothing") + + down.Store(false) + clock.Lock() + now = now.Add(breakerCool) + clock.Unlock() + c, err := m.Reserve(t.Context(), keys("a"), time.Minute) + require.NoError(t, err, "it closes after the cool-down") + assert.Equal(t, Claimed, c[0].Status) +} + +func TestBreaker_FailuresSpreadOutDoNotTrip(t *testing.T) { + t.Parallel() + now := time.Unix(1_000_000, 0) + b := newBreaker(func() time.Time { return now }) + fail := fmt.Errorf("%w: x", ErrUnavailable) + for range 3 * breakerTrips { + b.record(fail) + now = now.Add(breakerWindow/(breakerTrips-1) + time.Millisecond) + } + require.NoError(t, b.allow()) + now = now.Add(2 * breakerWindow) + for range breakerTrips - 1 { + b.record(fail) + } + b.record(nil) + b.record(fail) + require.NoError(t, b.allow(), "a success resets the count") +} + +func TestDynamo_Check(t *testing.T) { + t.Parallel() + good := &dynamodb.DescribeTableOutput{Table: &types.TableDescription{ + KeySchema: []types.KeySchemaElement{{AttributeName: aws.String("pk"), KeyType: types.KeyTypeHash}}, + AttributeDefinitions: []types.AttributeDefinition{{AttributeName: aws.String("pk"), AttributeType: types.ScalarAttributeTypeS}}, + }} + ttlOn := &dynamodb.DescribeTimeToLiveOutput{TimeToLiveDescription: &types.TimeToLiveDescription{ + AttributeName: aws.String("ex"), TimeToLiveStatus: types.TimeToLiveStatusEnabled, + }} + check := func(table *dynamodb.DescribeTableOutput, ttl *dynamodb.DescribeTimeToLiveOutput, ttlErr error) error { + f := &fakeDynamo{ + describe: func() (*dynamodb.DescribeTableOutput, error) { return table, nil }, + ttl: func() (*dynamodb.DescribeTimeToLiveOutput, error) { return ttl, ttlErr }, + } + return newDynamo(f, DynamoConfig{Table: "dedupe"}).Check(t.Context()) + } + require.NoError(t, check(good, ttlOn, nil)) + require.NoError(t, check(good, &dynamodb.DescribeTimeToLiveOutput{}, nil), "no TTL is a warning") + require.ErrorIs(t, check(good, nil, &types.InternalServerError{}), ErrUnavailable) + + withRange := &dynamodb.DescribeTableOutput{Table: &types.TableDescription{ + KeySchema: []types.KeySchemaElement{ + {AttributeName: aws.String("pk"), KeyType: types.KeyTypeHash}, + {AttributeName: aws.String("sk"), KeyType: types.KeyTypeRange}, + }, + }} + assert.ErrorContains(t, check(withRange, ttlOn, nil), "key schema") +} + +func TestDynamo_Config(t *testing.T) { + t.Parallel() + c := DynamoConfig{}.withDefaults() + assert.Equal(t, DynamoConfig{Timeout: 250 * time.Millisecond, MaxAttempts: 3, RetryMode: "standard", ReserveConcurrency: 64}, c) + for _, mode := range []string{"standard", "adaptive"} { + r, err := newRetryer(DynamoConfig{RetryMode: mode, MaxAttempts: 4}) + require.NoError(t, err) + assert.Equal(t, 4, r().MaxAttempts()) + } + _, err := newRetryer(DynamoConfig{RetryMode: "legacy"}) + require.Error(t, err) + + _, err = NewDynamo(t.Context(), DynamoConfig{}) + require.ErrorContains(t, err, "table is required") + _, err = NewDynamo(t.Context(), DynamoConfig{Table: "t", RetryMode: "legacy"}) + require.Error(t, err) + d, err := NewDynamo(t.Context(), DynamoConfig{Table: "t", Region: "us-east-1"}) + require.NoError(t, err) + require.ErrorIs(t, d.CreateTable(t.Context()), ErrCreateTableNeedsEndpoint, "never against real AWS") +} + +func TestFullJitter(t *testing.T) { + t.Parallel() + for _, tc := range []struct { + attempt int + most time.Duration + }{ + {-1, 10 * time.Millisecond}, + {0, 10 * time.Millisecond}, + {2, 40 * time.Millisecond}, + {3, 50 * time.Millisecond}, + {1 << 20, 50 * time.Millisecond}, + } { + seen := map[time.Duration]bool{} + for range 200 { + d := fullJitter(10*time.Millisecond, 50*time.Millisecond, tc.attempt) + assert.GreaterOrEqual(t, d, time.Duration(0)) + assert.LessOrEqual(t, d, tc.most, "attempt %d", tc.attempt) + seen[d] = true + } + assert.Greater(t, len(seen), 1, "attempt %d: jittered, not a fixed wait", tc.attempt) + } + assert.Zero(t, fullJitter(10*time.Millisecond, 0, 3)) +} + +// Whatever the Timeout and MaxAttempts, the SDK's retries of one call wait +// at most half the Timeout between them. +func TestRetryBackoff_FitsTheTimeout(t *testing.T) { + t.Parallel() + for _, cfg := range []DynamoConfig{ + {}, + {MaxAttempts: 10}, + {Timeout: 5 * time.Second, MaxAttempts: 2}, + {Timeout: 40 * time.Millisecond, MaxAttempts: 5}, + } { + cfg = cfg.withDefaults() + r, err := newRetryer(cfg) + require.NoError(t, err) + retryer := r() + var worst time.Duration + // The SDK numbers a call's retries from 1 (from 0 under its 2026 + // retry behaviour); the later ones are the longer. + for attempt := 1; attempt < cfg.MaxAttempts; attempt++ { + var most time.Duration + seen := map[time.Duration]bool{} + for range 500 { + d, err := retryer.RetryDelay(attempt, nil) + require.NoError(t, err) + most = max(most, d) + seen[d] = true + } + assert.Greater(t, len(seen), 1, "retries are jittered, not in lockstep") + worst += most + } + assert.LessOrEqual(t, worst, cfg.Timeout/2, "%+v", cfg) + } +} + +func TestDynamo_CommitBackoffIsJittered(t *testing.T) { + t.Parallel() + d := newDynamo(&fakeDynamo{}, DynamoConfig{Table: "dedupe"}) + for attempt := range commitRounds { + seen := map[time.Duration]bool{} + for range 200 { + w := d.commitBackoff(attempt) + assert.LessOrEqual(t, w, commitCeiling) + seen[w] = true + } + assert.Greater(t, len(seen), 1, "round %d", attempt) + } +} + +// throttledHTTP answers every request with a DynamoDB throttle, counting +// the PutItems. +type throttledHTTP struct{ puts atomic.Int64 } + +func (h *throttledHTTP) Do(r *http.Request) (*http.Response, error) { + if r.Header.Get("X-Amz-Target") == "DynamoDB_20120810.PutItem" { + h.puts.Add(1) + } + body := `{"__type":"com.amazonaws.dynamodb.v20120810#ProvisionedThroughputExceededException","message":"injected"}` + return &http.Response{ + StatusCode: http.StatusBadRequest, + Header: http.Header{"Content-Type": {"application/x-amz-json-1.0"}}, + Body: io.NopCloser(strings.NewReader(body)), + }, nil +} + +// Through the real SDK stack at the default Timeout and MaxAttempts: a +// throttled put is retried until its attempts run out, inside the call's +// deadline, so the Reserve fails with the throttle as its cause. +func TestDynamo_ThrottledCallEndsOnItsLastAttempt(t *testing.T) { + t.Parallel() + h := &throttledHTTP{} + d, err := NewDynamo(t.Context(), DynamoConfig{Table: "dedupe", Region: "us-east-1", Endpoint: "http://dynamodb.invalid"}, + config.WithCredentialsProvider(credentials.NewStaticCredentialsProvider("k", "s", "")), + config.WithHTTPClient(h)) + require.NoError(t, err) + m := d.Tenant("acme") + require.NoError(t, m.Apply(true)) + calls := breakerTrips - 1 + for range calls { + start := time.Now() + _, err := m.Reserve(t.Context(), keys("a"), time.Minute) + took := time.Since(start) + require.ErrorIs(t, err, ErrUnavailable) + var maxed *retry.MaxAttemptsError + require.ErrorAs(t, err, &maxed, "the attempts ran out, not the deadline") + var throttled *types.ProvisionedThroughputExceededException + require.ErrorAs(t, err, &throttled) + require.NotErrorIs(t, err, context.DeadlineExceeded) + assert.Less(t, took, d.cfg.Timeout) + } + assert.Equal(t, int64(calls*d.cfg.MaxAttempts), h.puts.Load()) +} + +// The idle pool holds a connection for every call one Reserve can have in +// flight, so a wide Reserve reuses them rather than dial; an HTTP client in +// extra (TestDynamo_ThrottledCallEndsOnItsLastAttempt's) replaces it. +func TestNewDynamo_SizesTheIdlePool(t *testing.T) { + t.Parallel() + for _, n := range []int{0, 4, 8, 200} { + d, err := NewDynamo(t.Context(), DynamoConfig{Table: "dedupe", Region: "us-east-1", ReserveConcurrency: n}) + require.NoError(t, err) + client, ok := d.api.(*dynamodb.Client).Options().HTTPClient.(*awshttp.BuildableClient) + require.True(t, ok) + tr := client.GetTransport() + assert.GreaterOrEqual(t, tr.MaxIdleConnsPerHost, d.cfg.ReserveConcurrency, "ReserveConcurrency %d", n) + assert.GreaterOrEqual(t, tr.MaxIdleConns, d.cfg.ReserveConcurrency, "ReserveConcurrency %d", n) + assert.GreaterOrEqual(t, tr.MaxIdleConnsPerHost, awshttp.DefaultHTTPTransportMaxIdleConnsPerHost, "never below the SDK's default: ReserveConcurrency %d", n) + assert.GreaterOrEqual(t, tr.MaxIdleConns, awshttp.DefaultHTTPTransportMaxIdleConns, "ReserveConcurrency %d", n) + } +} + +func TestExpiresAt(t *testing.T) { + t.Parallel() + base := time.Unix(100, 0) + assert.Equal(t, int64(101), expiresAt(base, time.Second)) + assert.Equal(t, int64(102), expiresAt(base, 1500*time.Millisecond), "rounded up: never ends early") + assert.Equal(t, int64(102), expiresAt(base.Add(time.Nanosecond), time.Second)) +} + +func idOf(av types.AttributeValue) string { + s := av.(*types.AttributeValueMemberS).Value + return s[len(s)-2:] +} + +func TestDynamo_CommitAttemptsEveryChunk(t *testing.T) { + t.Parallel() + var mu sync.Mutex + written := 0 + _, m := openFakeWith(t, &fakeDynamo{batch: func(_ context.Context, in *dynamodb.BatchWriteItemInput) (*dynamodb.BatchWriteItemOutput, error) { + reqs := in.RequestItems["dedupe"] + if idOf(reqs[0].PutRequest.Item[attrKey]) == "00" { + return nil, &types.InternalServerError{} + } + mu.Lock() + written += len(reqs) + mu.Unlock() + return &dynamodb.BatchWriteItemOutput{}, nil + }}, DynamoConfig{Table: "dedupe", ReserveConcurrency: 1}) + var claims []Claim + for i := range 3 * batchWriteMax { + claims = append(claims, Claim{Key: keys(fmt.Sprintf("%02d", i))[0], Status: Claimed, Token: "t"}) + } + require.ErrorIs(t, m.Commit(t.Context(), claims, 0), ErrUnavailable) + assert.Equal(t, 2*batchWriteMax, written, "a failed chunk does not cancel the others: their records are published") +} + +// A caller that cancels mid-Reserve (a client disconnecting) leaves nothing +// claimed. The fake applies a put after a delay whatever the caller does, as +// DynamoDB applies a request already on the wire, so a put abandoned on the +// cancel would land after its release and hold its id for the lease. No +// put applies before the cancel, so neither case depends on timing. +func TestDynamo_CallerCancelLeavesNothingClaimed(t *testing.T) { + t.Parallel() + t.Run("a put still unsent", func(t *testing.T) { + t.Parallel() + assertCancelLeavesNothing(t, keys("k0", "k1", "k2")) + }) + t.Run("every put sent", func(t *testing.T) { + t.Parallel() + assertCancelLeavesNothing(t, keys("k0", "k1")) + }) +} + +// assertCancelLeavesNothing cancels a Reserve of ks once two puts are sent. +func assertCancelLeavesNothing(t *testing.T, ks []Key) { + t.Helper() + var ( + mu sync.Mutex + table = map[string]string{} + sent []string + applies sync.WaitGroup + ) + started := make(chan struct{}, 2) + cancelled := make(chan struct{}) + _, m := openFakeWith(t, &fakeDynamo{ + put: func(ctx context.Context, in *dynamodb.PutItemInput) (*dynamodb.PutItemOutput, error) { + id := idOf(in.Item[attrKey]) + tk := string(in.Item[attrToken].(*types.AttributeValueMemberB).Value) + mu.Lock() + sent = append(sent, id) + mu.Unlock() + applied := make(chan struct{}) + applies.Go(func() { + <-cancelled + time.Sleep(5 * time.Millisecond) // lands after the caller left + mu.Lock() + table[id] = tk + mu.Unlock() + close(applied) + }) + started <- struct{}{} + select { + case <-applied: + return &dynamodb.PutItemOutput{}, nil + case <-ctx.Done(): + return nil, ctx.Err() + } + }, + del: func(_ context.Context, in *dynamodb.DeleteItemInput) (*dynamodb.DeleteItemOutput, error) { + id := idOf(in.Key[attrKey]) + tk := string(in.ExpressionAttributeValues[":tk"].(*types.AttributeValueMemberB).Value) + mu.Lock() + defer mu.Unlock() + if table[id] != tk { + return nil, &types.ConditionalCheckFailedException{} + } + delete(table, id) + return &dynamodb.DeleteItemOutput{}, nil + }, + }, DynamoConfig{Table: "dedupe", ReserveConcurrency: 2}) + ctx, cancel := context.WithCancel(t.Context()) + go func() { <-started; <-started; cancel(); close(cancelled) }() + _, err := m.Reserve(ctx, ks, time.Minute) + require.ErrorIs(t, err, context.Canceled) + applies.Wait() + mu.Lock() + defer mu.Unlock() + assert.Len(t, sent, 2, "a put not yet sent is skipped") + assert.Empty(t, table, "every applied put was released") +} + +func TestDynamo_ReleaseAttemptsEveryClaim(t *testing.T) { + t.Parallel() + var deletes atomic.Int64 + _, m := openFakeWith(t, &fakeDynamo{del: func(_ context.Context, in *dynamodb.DeleteItemInput) (*dynamodb.DeleteItemOutput, error) { + deletes.Add(1) + if idOf(in.Key[attrKey]) == "k0" { + return nil, &types.ProvisionedThroughputExceededException{} + } + return &dynamodb.DeleteItemOutput{}, nil + }}, DynamoConfig{Table: "dedupe", ReserveConcurrency: 1}) + var claims []Claim + for _, k := range keys("k0", "k1", "k2", "k3") { + claims = append(claims, Claim{Key: k, Status: Claimed, Token: "t"}) + } + require.ErrorIs(t, m.Release(t.Context(), claims), ErrUnavailable) + assert.Equal(t, int64(4), deletes.Load()) +} + +// One throttled put in a multi-key Reserve: the unsent puts are never sent, +// and a sibling already sent runs to its answer before the undo releases it, +// so a sibling's failure never lets a put land after its own release. +func TestDynamo_FailedMultiKeyReserve(t *testing.T) { + t.Parallel() + var mu sync.Mutex + var put, released []string + landed := map[string]bool{} + _, m := openFakeWith(t, &fakeDynamo{ + put: func(ctx context.Context, in *dynamodb.PutItemInput) (*dynamodb.PutItemOutput, error) { + id := idOf(in.Item[attrKey]) + mu.Lock() + put = append(put, id) + mu.Unlock() + if id == "k0" { + return nil, &types.ProvisionedThroughputExceededException{} + } + time.Sleep(20 * time.Millisecond) + if err := ctx.Err(); err != nil { + return nil, err + } + mu.Lock() + landed[id] = true + mu.Unlock() + return &dynamodb.PutItemOutput{}, nil + }, + del: func(_ context.Context, in *dynamodb.DeleteItemInput) (*dynamodb.DeleteItemOutput, error) { + id := idOf(in.Key[attrKey]) + mu.Lock() + defer mu.Unlock() + released = append(released, id) + if id != "k0" && !landed[id] { + t.Errorf("%s released before its put answered", id) + } + return &dynamodb.DeleteItemOutput{}, nil + }, + }, DynamoConfig{Table: "dedupe", ReserveConcurrency: 2}) + ks := keys("k0", "k1", "k2", "k3", "k4", "k5", "k6", "k7") + _, err := m.Reserve(t.Context(), ks, time.Minute) + require.ErrorIs(t, err, ErrUnavailable) + mu.Lock() + defer mu.Unlock() + assert.ElementsMatch(t, put, released, "exactly the sent puts are released") + assert.Less(t, len(put), len(ks), "unsent puts were never sent") +} diff --git a/internal/dedupe/embedded.go b/internal/dedupe/embedded.go index 6b3b5efaf..227564b71 100644 --- a/internal/dedupe/embedded.go +++ b/internal/dedupe/embedded.go @@ -4,9 +4,13 @@ import ( "context" "encoding/binary" "errors" + "fmt" + "hash/fnv" "math" "path/filepath" + "strconv" "sync" + "sync/atomic" "time" "github.com/cockroachdb/pebble" @@ -16,23 +20,55 @@ import ( // Embedded is the embedded implementation: every tenant's seen ids in one // Pebble instance at data_dir/pebble, each key led by its tenant (#583 story -// 3), so a thousand tenants cost one instance's goroutines, open files and -// heap rather than a thousand. The instance opens with the first tenant's -// store switched on and closes with the last one switched off: it is open -// exactly while some tenant has dedupe on, and a tenant switched off, -// rejected or removed keeps its seen ids for when it is back. +// 3) and then its table (#222), with the pending claims in memory beside it +// (pendingSet). One instance means a thousand tenants cost one instance's +// goroutines, open files and heap rather than a thousand. The instance opens +// with the first tenant's store switched on and closes with the last one +// switched off: it is open exactly while some tenant has dedupe on, and a +// tenant switched off, rejected or removed keeps its seen ids for when it is +// back. Pebble is one process's, so two pods on it do not share seen ids. type Embedded struct { dir string - mu sync.Mutex // guards db and open - db *pebble.DB - open int // tenant stores open over db + mu sync.Mutex // guards db, open and stopSweep + db *pebble.DB + open int // tenant stores open over db + stopSweep func() // stops db's sweep + + // commitMu is read-held by Commit and held by a sweep chunk while it + // re-reads and deletes, so a sweep never deletes a key a Commit rewrote + // after the sweep read it. + commitMu sync.RWMutex + sweepFirst time.Duration + sweepEvery time.Duration + + pending *pendingSet + tokens atomic.Uint64 + now func() time.Time + // readHook, when set, runs before each Pebble read in Reserve; a test + // makes it fail to exercise Reserve's all-or-nothing error path. + readHook func() error + // sweepScanHook and sweepDeleteHook, when set, run in a sweep chunk: + // between its unlocked read and its re-read, and between its re-read and + // its delete. A test races a Commit into each gap. sweepReadHook, when + // set, runs once per key sweepCandidates' unlocked read visits, from + // inside its loop — a test TryLocks commitMu there to prove the read + // itself never holds it, not just the instant after it returns. + sweepScanHook func() + sweepDeleteHook func() + sweepReadHook func() } // NewEmbedded returns the embedded implementation under dataDir. Nothing is // opened until a tenant's store is. func NewEmbedded(dataDir string) *Embedded { - return &Embedded{dir: filepath.Join(dataDir, "pebble")} + return &Embedded{ + dir: filepath.Join(dataDir, "pebble"), + pending: newPendingSet(), + now: time.Now, + sweepFirst: sweepFirstDelay, + sweepEvery: sweepInterval, + } } // Dir is where the instance lives. @@ -45,14 +81,10 @@ func (e *Embedded) Open() bool { return e.db != nil } -// keySeparator ends the tenant at the front of every key. A tenant id has no -// NUL, so the first one in a key is this one, and no two tenants' keys meet. -const keySeparator = 0 - // Tenant builds tenant id's store, closed, over its share of the instance — // the Factory Stores takes. func (e *Embedded) Tenant(id tenant.ID) *Managed { - prefix := append([]byte(id), keySeparator) + prefix := KeyPrefix(id) return NewManaged(func() (Deduplicator, error) { return e.acquire(prefix) }) } @@ -67,6 +99,7 @@ func (e *Embedded) acquire(prefix []byte) (Deduplicator, error) { return nil, err } e.db = db + e.stopSweep = e.startSweep(db) } e.open++ return &tenantStore{e: e, db: e.db, prefix: prefix}, nil @@ -81,6 +114,7 @@ func (e *Embedded) release() error { if e.open > 0 { return nil } + e.stopSweep() err := e.db.Close() e.db = nil return err @@ -115,28 +149,147 @@ type tenantStore struct { closed sync.Once } -// CheckAndMark returns true if the event was already seen. -func (s *tenantStore) CheckAndMark(_ context.Context, eventID string) (bool, error) { - key := make([]byte, 0, len(s.prefix)+len(eventID)) - key = append(append(key, s.prefix...), eventID...) +// Committed values are committedMark ‖ expiry (big-endian UnixNano, 0 = +// never). Values written before #222 were a bare 8-byte timestamp, so one +// under a key that happens to equal a current one reads as absent. +const ( + committedMark = 2 + valueLen = 9 +) - _, closer, err := s.db.Get(key) - if err == nil { +// Reserve claims each key under its shard's lock: the pending check, the +// Pebble read and the claim happen with no other Reserve for that key in +// between, and Pebble's directory lock keeps a second process off the +// instance, so at most one caller holds a key (#390). +func (s *tenantStore) Reserve(_ context.Context, keys []Key, lease time.Duration) ([]Claim, error) { + now := s.e.now() + claims := make([]Claim, 0, len(keys)) + for _, k := range keys { + c, err := s.reserve(AppendKey(nil, s.prefix, k), k, now, lease) + if err != nil { + s.release(claims) + return nil, err + } + claims = append(claims, c) + } + return claims, nil +} + +func (s *tenantStore) reserve(key []byte, k Key, now time.Time, lease time.Duration) (Claim, error) { + sh := s.e.pending.shard(key) + sh.mu.Lock() + defer sh.mu.Unlock() + sh.sweep(now) + if p, ok := sh.m[string(key)]; ok && now.Before(p.expires) { + return Claim{Key: k, Status: InFlight}, nil + } + if s.e.readHook != nil { + if err := s.e.readHook(); err != nil { + return Claim{}, err + } + } + val, closer, err := s.db.Get(key) + switch { + case err == nil: + live := committedLive(val, now) _ = closer.Close() - return true, nil + if live { + return Claim{Key: k, Status: Duplicate}, nil + } + case !errors.Is(err, pebble.ErrNotFound): + return Claim{}, fmt.Errorf("dedupe read: %w", err) + } + token := strconv.FormatUint(s.e.tokens.Add(1), 36) + sh.m[string(key)] = pending{token: token, expires: now.Add(lease)} + return Claim{Key: k, Status: Claimed, Token: token}, nil +} + +// committedLive reports whether a stored value is a commit that has not +// expired. +func committedLive(val []byte, now time.Time) bool { + exp, ok := committedExpiry(val) + return ok && (exp == 0 || now.UnixNano() < exp) +} + +// committedExpired reports whether a stored value is a commit whose +// retention has ended — what the sweep deletes. +func committedExpired(val []byte, now time.Time) bool { + exp, ok := committedExpiry(val) + return ok && exp != 0 && now.UnixNano() >= exp +} + +// isCommit reports whether val is a commit this layout wrote. +func isCommit(val []byte) bool { + _, ok := committedExpiry(val) + return ok +} + +// committedExpiry reads a commit's expiry (UnixNano, 0 = never); ok is false +// for a value that is not a commit. +func committedExpiry(val []byte) (exp int64, ok bool) { + if len(val) != valueLen || val[0] != committedMark { + return 0, false + } + return int64(binary.BigEndian.Uint64(val[1:])), true //nolint:gosec // written from an int64 below +} + +// Commit writes every claim in one batch and one fsync, then drops the +// pending entries it still owns — in that order, so no Reserve in between +// finds the key neither pending nor committed. +func (s *tenantStore) Commit(_ context.Context, claims []Claim, retention time.Duration) error { + s.e.commitMu.RLock() + defer s.e.commitMu.RUnlock() + exp := expiry(s.e.now(), retention) + val := make([]byte, valueLen) + val[0] = committedMark + binary.BigEndian.PutUint64(val[1:], uint64(exp)) //nolint:gosec // expiry is never negative + b := s.db.NewBatch() + defer func() { _ = b.Close() }() + for _, c := range claims { + if err := b.Set(AppendKey(nil, s.prefix, c.Key), val, nil); err != nil { + return fmt.Errorf("dedupe commit: %w", err) + } } - if !errors.Is(err, pebble.ErrNotFound) { - return false, err + if err := b.Commit(pebble.Sync); err != nil { + return fmt.Errorf("dedupe commit: %w", err) } + s.release(claims) + return nil +} - // Store timestamp as value for future auditing. - val := make([]byte, 8) - binary.BigEndian.PutUint64(val, uint64(time.Now().UnixNano())) +// expiry is the stored expiry of a commit at now kept for retention: 0 for +// none, and the latest representable instant for a retention reaching past +// it, rather than a wrapped-around one in the past. +func expiry(now time.Time, retention time.Duration) int64 { + if retention <= 0 { + return 0 + } + n := now.UnixNano() + if retention > time.Duration(math.MaxInt64-n) { + return math.MaxInt64 + } + return n + int64(retention) +} - if err := s.db.Set(key, val, pebble.Sync); err != nil { - return false, err +// Release drops the pending entries the claims still own. +func (s *tenantStore) Release(_ context.Context, claims []Claim) error { + s.release(claims) + return nil +} + +func (s *tenantStore) release(claims []Claim) { + for _, c := range claims { + if c.Status != Claimed { + continue + } + key := AppendKey(nil, s.prefix, c.Key) + sh := s.e.pending.shard(key) + sh.mu.Lock() + if p, ok := sh.m[string(key)]; ok && p.token == c.Token { + delete(sh.m, string(key)) + } + sh.mu.Unlock() } - return false, nil } // Close releases the store's hold on the instance. Safe to call more than @@ -146,3 +299,51 @@ func (s *tenantStore) Close() error { s.closed.Do(func() { err = s.e.release() }) return err } + +// pendingShards spreads the pending claims over independently locked maps, +// so Reserves for different keys rarely wait on each other. +const pendingShards = 64 + +type pending struct { + token string + expires time.Time +} + +type pendingShard struct { + mu sync.Mutex + m map[string]pending + nextSweep time.Time +} + +// sweep drops lapsed claims at most once a DefaultLease, so a claim nobody +// commits, releases or re-reserves does not stay in memory. Callers hold mu. +func (sh *pendingShard) sweep(now time.Time) { + if now.Before(sh.nextSweep) { + return + } + sh.nextSweep = now.Add(DefaultLease) + for k, p := range sh.m { + if !now.Before(p.expires) { + delete(sh.m, k) + } + } +} + +// pendingSet is every tenant's live claims. It lives in memory because one +// process owns the instance: a crash forgets every claim, which is each +// lease lapsing at once. +type pendingSet [pendingShards]pendingShard + +func newPendingSet() *pendingSet { + p := new(pendingSet) + for i := range p { + p[i].m = map[string]pending{} + } + return p +} + +func (p *pendingSet) shard(key []byte) *pendingShard { + h := fnv.New32a() + _, _ = h.Write(key) + return &p[h.Sum32()%pendingShards] +} diff --git a/internal/dedupe/embedded_test.go b/internal/dedupe/embedded_test.go index de65a1512..3bf9ec2f2 100644 --- a/internal/dedupe/embedded_test.go +++ b/internal/dedupe/embedded_test.go @@ -2,8 +2,13 @@ package dedupe import ( "context" + "maps" "os" + "slices" "testing" + "time" + + "github.com/cockroachdb/pebble" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -26,15 +31,15 @@ func TestEmbedded_FirstSeenThenDuplicate(t *testing.T) { m := switchedOn(t, NewEmbedded(t.TempDir()), "acme") ctx := context.Background() - dup, err := m.CheckAndMark(ctx, "event-1") + dup, err := mark(ctx, m, "event-1") require.NoError(t, err) assert.False(t, dup, "first occurrence must not be a duplicate") - dup, err = m.CheckAndMark(ctx, "event-1") + dup, err = mark(ctx, m, "event-1") require.NoError(t, err) assert.True(t, dup, "second occurrence of the same id must be a duplicate") - dup, err = m.CheckAndMark(ctx, "event-2") + dup, err = mark(ctx, m, "event-2") require.NoError(t, err) assert.False(t, dup, "distinct ids are independent") } @@ -49,16 +54,16 @@ func TestEmbedded_TenantsDoNotShareSeenIDs(t *testing.T) { ctx := context.Background() a, ab := switchedOn(t, e, "a"), switchedOn(t, e, "ab") - dup, err := a.CheckAndMark(ctx, "bc") + dup, err := mark(ctx, a, "bc") require.NoError(t, err) assert.False(t, dup) - dup, err = ab.CheckAndMark(ctx, "c") + dup, err = mark(ctx, ab, "c") require.NoError(t, err) assert.False(t, dup, "another tenant's key, however the two would join") - dup, err = ab.CheckAndMark(ctx, "bc") + dup, err = mark(ctx, ab, "bc") require.NoError(t, err) assert.False(t, dup, "an id tenant a has seen is new to tenant ab") - dup, err = a.CheckAndMark(ctx, "bc") + dup, err = mark(ctx, a, "bc") require.NoError(t, err) assert.True(t, dup, "and still a duplicate within its own tenant") } @@ -77,7 +82,7 @@ func TestEmbedded_OpenWhileAnyTenantStoreIs(t *testing.T) { require.NoError(t, acme.Apply(true)) require.NoError(t, globex.Apply(true)) assert.True(t, e.Open()) - _, err := acme.CheckAndMark(ctx, "e1") + _, err := mark(ctx, acme, "e1") require.NoError(t, err) require.NoError(t, acme.Apply(false)) @@ -90,7 +95,7 @@ func TestEmbedded_OpenWhileAnyTenantStoreIs(t *testing.T) { require.NoError(t, acme.Apply(true)) t.Cleanup(func() { _ = acme.Close() }) - dup, err := acme.CheckAndMark(ctx, "e1") + dup, err := mark(ctx, acme, "e1") require.NoError(t, err) assert.True(t, dup, "a tenant switched off keeps its seen ids") } @@ -104,7 +109,7 @@ func TestEmbedded_StatsAreTheInstances(t *testing.T) { acme := switchedOn(t, e, "acme") switchedOn(t, e, "globex") - _, err := acme.CheckAndMark(context.Background(), "e1") + _, err := mark(context.Background(), acme, "e1") require.NoError(t, err) stats := e.Stats() m := e.db.Metrics() @@ -126,7 +131,7 @@ func TestEmbedded_OpenFailure(t *testing.T) { require.Error(t, acme.Apply(true)) require.Error(t, globex.Apply(true), "one instance: its failure is every tenant's") assert.False(t, e.Open()) - _, err := acme.CheckAndMark(context.Background(), "e1") + _, err := mark(context.Background(), acme, "e1") require.ErrorIs(t, err, ErrUnavailable) require.NoError(t, os.Remove(e.Dir())) @@ -134,3 +139,86 @@ func TestEmbedded_OpenFailure(t *testing.T) { t.Cleanup(func() { _ = acme.Close() }) assert.True(t, e.Open()) } + +// Keys from before the table joined the key (#222) never count: an id seen +// then is accepted once more after the upgrade, the documented cost of the +// new layout. A tenant ‖ NUL ‖ id key is never looked up; a bare v0.1.0 id +// that spells a current key is, and its 8-byte value reads as absent. +func TestEmbedded_VersionZeroKeysDoNotCount(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + m := switchedOn(t, e, "acme") + require.NoError(t, e.db.Set([]byte("acme\x00e1"), make([]byte, 8), pebble.Sync)) + stale := AppendKey(nil, KeyPrefix("acme"), Key{Table: "events", ID: "e2"}) + require.NoError(t, e.db.Set(stale, make([]byte, 8), pebble.Sync)) + for _, id := range []string{"e1", "e2"} { + dup, err := mark(context.Background(), m, id) + require.NoError(t, err) + assert.False(t, dup, id) + } + dup, err := mark(context.Background(), m, "e2") + require.NoError(t, err) + assert.True(t, dup, "the commit overwrote the stale value") +} + +// v0.1.0 stored a bare id as the key with an 8-byte value, so a v0.1.0 id +// that happens to spell a key the current layout would also write — +// tenant "0", table "events", id "e1" join to "0/events/e1", which a v0.1.0 +// record could have used as its own id — must not read as a live duplicate: +// only a value of exactly valueLen bytes leading with committedMark is ours. +// The planted value leads with committedMark, so only the length check can +// refuse it (and keeps a short value from being decoded as an expiry). +// The reserve below claims the key despite the stale value, and only the +// commit it makes turns a second reserve of the same id into a duplicate. +func TestEmbedded_PreJoinValueUnderACollidingKeyIsNotLive(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + m := switchedOn(t, e, "0") + key := AppendKey(nil, KeyPrefix("0"), Key{Table: "events", ID: "e1"}) + stale := make([]byte, 8) + stale[0] = committedMark + require.NoError(t, e.db.Set(key, stale, pebble.Sync)) + + dup, err := mark(context.Background(), m, "e1") + require.NoError(t, err) + assert.False(t, dup, "an 8-byte v0.1.0 value is not this layout's commit") + + dup, err = mark(context.Background(), m, "e1") + require.NoError(t, err) + assert.True(t, dup, "the mark above overwrote it with a real commit") +} + +// Same colliding key, a 9-byte value that committedMark did not write: the +// mark byte, not just the length, is what says a value is ours. +func TestEmbedded_WrongMarkByteIsNotLive(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + m := switchedOn(t, e, "0") + key := AppendKey(nil, KeyPrefix("0"), Key{Table: "events", ID: "e1"}) + val := make([]byte, valueLen) + val[0] = committedMark + 1 + require.NoError(t, e.db.Set(key, val, pebble.Sync)) + + dup, err := mark(context.Background(), m, "e1") + require.NoError(t, err) + assert.False(t, dup, "a value not led by committedMark is not a live commit") +} + +// A claim nobody commits, releases or reserves again leaves memory at the +// next sweep past its lease, not never. +func TestPendingShard_SweepDropsLapsedClaims(t *testing.T) { + t.Parallel() + now := time.Now() + sh := &pendingShard{m: map[string]pending{ + "lapsed": {token: "1", expires: now.Add(-time.Second)}, + "live": {token: "2", expires: now.Add(time.Hour)}, + }} + sh.sweep(now) + assert.Equal(t, []string{"live"}, slices.Collect(maps.Keys(sh.m))) + + sh.m["lapsed"] = pending{token: "3", expires: now.Add(-time.Second)} + sh.sweep(now.Add(time.Second)) + assert.Len(t, sh.m, 2, "at most one sweep per DefaultLease") + sh.sweep(now.Add(DefaultLease)) + assert.Len(t, sh.m, 1) +} diff --git a/internal/dedupe/export_test.go b/internal/dedupe/export_test.go new file mode 100644 index 000000000..76e9dbc36 --- /dev/null +++ b/internal/dedupe/export_test.go @@ -0,0 +1,39 @@ +package dedupe + +import ( + "context" + "errors" + "sync/atomic" + "time" +) + +// SetClock replaces e's clock, for tests that let leases and retentions lapse +// without sleeping. +func SetClock(e *Embedded, now func() time.Time) { e.now = now } + +// FailNextReserve makes e's next Reserve fail after it has claimed n keys, +// once. +func FailNextReserve(e *Embedded, n int) { + var reads atomic.Int64 + var failed atomic.Bool + e.readHook = func() error { + if reads.Add(1) > int64(n) && failed.CompareAndSwap(false, true) { + return errors.New("injected read failure") + } + return nil + } +} + +// mark reserves and commits id in table "events", reporting whether it was +// already committed — the old check-and-mark, for tests about everything +// else. +func mark(ctx context.Context, d Deduplicator, id string) (bool, error) { + claims, err := d.Reserve(ctx, []Key{{Table: "events", ID: id}}, DefaultLease) + if err != nil { + return false, err + } + if claims[0].Status != Claimed { + return true, nil + } + return false, d.Commit(ctx, claims, 0) +} diff --git a/internal/dedupe/key.go b/internal/dedupe/key.go new file mode 100644 index 000000000..cd6f4b24a --- /dev/null +++ b/internal/dedupe/key.go @@ -0,0 +1,67 @@ +package dedupe + +import ( + "crypto/sha256" + "encoding/hex" + + "github.com/Wave-RF/WaveHouse/internal/keyenc" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// The key every backend stores is text: +// +// /
/ acme/clicks/evt-123 +// /
/# an id too long to store verbatim +// +// The table and id are escaped and joined by internal/keyenc, which never +// writes '/' or '#', and a tenant id holds neither (tenant.Parse), so the +// fields split back apart, a table name may hold any byte, and no two +// (tenant, table, id) triples share a key. A key is ASCII, so it is a valid +// DynamoDB String, and holds no NUL, so it never meets a tenant ‖ NUL ‖ id +// key written before #222. +const ( + keySep = '/' + hashedMark = '#' + // MaxIDBytes is the longest escaped id stored verbatim: a DynamoDB + // partition key holds at most 2,048 bytes, and the tenant and table share + // them. + MaxIDBytes = 1024 +) + +// KeyPrefix is the part of every key that names tenant id, so a backend +// computes it once per tenant store. +func KeyPrefix(id tenant.ID) []byte { + return append([]byte(id), keySep) +} + +// Hashed reports whether k's id is stored as its SHA-256 rather than +// verbatim: whether its escaped form is longer than MaxIDBytes. +func (k Key) Hashed() bool { + switch { + case len(k.ID) > MaxIDBytes: + return true + case 3*len(k.ID) <= MaxIDBytes: // escaping at most triples a byte + return false + } + return len(keyenc.Escape(k.ID)) > MaxIDBytes +} + +// AppendKey appends k's stored form, under the tenant prefix from KeyPrefix, +// to dst. +func AppendKey(dst, prefix []byte, k Key) []byte { + dst = append(dst, prefix...) + if k.Hashed() { + sum := sha256.Sum256([]byte(k.ID)) + dst = keyenc.AppendJoin(dst, keySep, k.Table) + return hex.AppendEncode(append(dst, keySep, hashedMark), sum[:]) + } + return keyenc.AppendJoin(dst, keySep, k.Table, k.ID) +} + +// IdempotencyKey is k's message id for the queue under tenant id: the first +// 128 bits of the stored key's SHA-256, in hex, so a republished record is +// recognised without its id riding in a header verbatim. +func IdempotencyKey(id tenant.ID, k Key) string { + sum := sha256.Sum256(AppendKey(nil, KeyPrefix(id), k)) + return hex.EncodeToString(sum[:16]) +} diff --git a/internal/dedupe/key_layout_test.go b/internal/dedupe/key_layout_test.go new file mode 100644 index 000000000..d7977f5cd --- /dev/null +++ b/internal/dedupe/key_layout_test.go @@ -0,0 +1,129 @@ +package dedupe_test + +import ( + "crypto/sha256" + "encoding/hex" + "strings" + "testing" + "unicode/utf8" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/keyenc" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +func key(tn tenant.ID, k dedupe.Key) string { + return string(dedupe.AppendKey(nil, dedupe.KeyPrefix(tn), k)) +} + +// The layout is pinned byte for byte: DynamoDB items and Pebble keys outlive +// the binary that wrote them. +func TestKeyLayout(t *testing.T) { + t.Parallel() + for _, tt := range []struct { + tn tenant.ID + k dedupe.Key + want string + }{ + {"acme", dedupe.Key{Table: "clicks", ID: "evt-123"}, "acme/clicks/evt-123"}, + {"acme-co", dedupe.Key{Table: "db.t", ID: "a/b"}, "acme-co/db%2Et/a%2Fb"}, + {"0", dedupe.Key{Table: "a\x00b", ID: "e1"}, "0/a%00b/e1"}, + {"0", dedupe.Key{Table: "", ID: ""}, "0//"}, + {"0", dedupe.Key{Table: "t", ID: "#%café"}, "0/t/%23%25caf%C3%A9"}, + } { + assert.Equal(t, tt.want, key(tt.tn, tt.k), "%+v", tt.k) + } + + long := strings.Repeat("x", dedupe.MaxIDBytes+1) + sum := sha256.Sum256([]byte(long)) + assert.Equal(t, "acme/t/#"+hex.EncodeToString(sum[:]), key("acme", dedupe.Key{Table: "t", ID: long})) +} + +// The limit is on the escaped id: 1,024 bytes that escape to more are hashed, +// and an id escaping to exactly 1,024 is not. +func TestKeyLayout_HashesOnTheEscapedLength(t *testing.T) { + t.Parallel() + fits := strings.Repeat("x", dedupe.MaxIDBytes) + assert.False(t, dedupe.Key{ID: fits}.Hashed()) + assert.Equal(t, "a/t/"+fits, key("a", dedupe.Key{Table: "t", ID: fits})) + + escapedFits := strings.Repeat(".", dedupe.MaxIDBytes/3) + "x" // 1,023 + 1 bytes escaped + assert.False(t, dedupe.Key{ID: escapedFits}.Hashed()) + escapedOver := strings.Repeat(".", dedupe.MaxIDBytes/3+1) // 1,026 bytes escaped + assert.True(t, dedupe.Key{ID: escapedOver}.Hashed()) + assert.True(t, strings.HasPrefix(key("a", dedupe.Key{Table: "t", ID: escapedOver}), "a/t/#")) +} + +// decodeKey inverts AppendKey. That it exists — every key splits back into +// the one triple that wrote it — is what makes the layout collision-free. +func decodeKey(t *testing.T, s string) (tn, table, idPart string) { + t.Helper() + require.True(t, utf8.ValidString(s), "a key is a valid DynamoDB String") + require.NotContains(t, s, "\x00") + parts, err := keyenc.Split(s, '/') + require.NoError(t, err) + require.Len(t, parts, 3, s) + _, err = tenant.Parse(parts[0]) + require.NoError(t, err) + return parts[0], parts[1], parts[2] +} + +// Triples a separator could confuse — the separator, the escape and hash +// marks, NUL, or what an escape looks like, in the table or the id, at either +// end, or moved across the table/id boundary — each get a key of their own +// and parse back to themselves. +func TestKeyLayout_NoCollisions(t *testing.T) { + t.Parallel() + tenants := []tenant.ID{"a", "ab", "a_b", "a-b", "acme"} + pieces := []string{"", "/", "%", "#", "\x00", "\xff", "a", "b", "a/", "/b", "a/b", "%2F", "a%2Fb", "#a", "acme/a", "é"} + seen := map[string]string{} + check := func(tn tenant.ID, k dedupe.Key) { + s := key(tn, k) + who := string(tn) + " | " + k.Table + " | " + k.ID + if prev, ok := seen[s]; ok && prev != who { + t.Fatalf("%q and %q share the key %q", prev, who, s) + } + seen[s] = who + gotTenant, gotTable, idPart := decodeKey(t, s) + assert.Equal(t, string(tn), gotTenant) + assert.Equal(t, k.Table, gotTable) + if !k.Hashed() { + assert.Equal(t, k.ID, idPart) + } + } + for _, tn := range tenants { + for _, table := range pieces { + for _, id := range pieces { + check(tn, dedupe.Key{Table: table, ID: id}) + } + } + } + // Every table and id up to three bytes over an alphabet of the bytes a + // layout could misread. + var all []string + var grow func(prefix string) + grow = func(prefix string) { + all = append(all, prefix) + if len(prefix) < 3 { + for _, c := range []string{"/", "%", "#", "2", "F", "\x00", "a"} { + grow(prefix + c) + } + } + } + grow("") + for _, tn := range tenants[:2] { + for _, table := range all { + for _, id := range all { + check(tn, dedupe.Key{Table: table, ID: id}) + } + } + } + // A hashed id never reads as a verbatim one, even one spelling the hash. + long := strings.Repeat("x", dedupe.MaxIDBytes+1) + sum := sha256.Sum256([]byte(long)) + check("a", dedupe.Key{Table: "t", ID: long}) + check("a", dedupe.Key{Table: "t", ID: "#" + hex.EncodeToString(sum[:])}) +} diff --git a/internal/dedupe/key_test.go b/internal/dedupe/key_test.go new file mode 100644 index 000000000..2f2981e22 --- /dev/null +++ b/internal/dedupe/key_test.go @@ -0,0 +1,33 @@ +package dedupe_test + +import ( + "strings" + "testing" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/stretchr/testify/assert" +) + +// The idempotency key is 32 hex characters, stable for one tenant, table and +// id, and different when any of the three differs — ids too long to store +// verbatim included. +func TestIdempotencyKey(t *testing.T) { + t.Parallel() + long := strings.Repeat("x", dedupe.MaxIDBytes+1) + base := dedupe.IdempotencyKey("acme", dedupe.Key{Table: "clicks", ID: "e1"}) + assert.Regexp(t, `^[0-9a-f]{32}$`, base) + assert.Equal(t, base, dedupe.IdempotencyKey("acme", dedupe.Key{Table: "clicks", ID: "e1"})) + + others := []string{ + dedupe.IdempotencyKey("globex", dedupe.Key{Table: "clicks", ID: "e1"}), + dedupe.IdempotencyKey("acme", dedupe.Key{Table: "views", ID: "e1"}), + dedupe.IdempotencyKey("acme", dedupe.Key{Table: "clicks", ID: "e2"}), + dedupe.IdempotencyKey("acme", dedupe.Key{Table: "clicks", ID: long}), + dedupe.IdempotencyKey("acme", dedupe.Key{Table: "clicks", ID: long + "y"}), + } + seen := map[string]bool{base: true} + for _, k := range others { + assert.False(t, seen[k], "collision: %s", k) + seen[k] = true + } +} diff --git a/internal/dedupe/managed.go b/internal/dedupe/managed.go index 3337e610d..09712c50b 100644 --- a/internal/dedupe/managed.go +++ b/internal/dedupe/managed.go @@ -3,26 +3,40 @@ package dedupe import ( "context" "errors" + "fmt" "sync" + "time" + + "go.opentelemetry.io/otel" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/metric" ) -// ErrDisabled is returned by Managed.CheckAndMark while dedupe is switched -// off. The ingest handler consults the settings snapshot before calling, so -// it only sees this in the window of a reload that flips dedupe.enabled: -// the snapshot and the store transition at different instants, and a record +// ErrDisabled is returned by Managed's calls while dedupe is switched off. +// The ingest handler consults the settings snapshot before calling, so it +// only sees this in the window of a reload that flips dedupe.enabled: the +// snapshot and the store transition at different instants, and a record // caught between them is published un-deduped rather than failed. var ErrDisabled = errors.New("dedupe is disabled") -// ErrUnavailable is returned by Managed.CheckAndMark when dedupe is switched -// on but the store failed to open. Ingest fails closed on it — the settings -// asked for dedupe, so publishing un-deduped is not a fallback. -var ErrUnavailable = errors.New("dedupe store is not open") +// ErrUnavailable is returned by Managed's calls when dedupe is switched on +// but the store failed to open, and wrapped by a backend's error when a +// retry later can succeed. Ingest fails closed on it — the settings asked +// for dedupe, so publishing un-deduped is not a fallback. +var ErrUnavailable = errors.New("dedupe store unavailable") + +// hashedIDCounter counts ids stored as their SHA-256 (Key.Hashed): an id +// longer than MaxIDBytes is a producer sending something other than an id. +var hashedIDCounter, _ = otel.Meter("wavehouse-dedupe").Int64Counter( + "wavehouse_dedupe_hashed_id_total", + metric.WithDescription("Dedupe ids stored as their SHA-256 because they exceed the verbatim length limit"), +) // Managed is a Deduplicator whose backing store follows the hot-reloadable // dedupe.enabled setting: Apply(true) opens it through the function -// NewManaged was given, Apply(false) closes it, and in-flight CheckAndMark -// calls are serialized against that swap so a reload can never close the -// store under a lookup. Which store that is — a tenant's share of the +// NewManaged was given, Apply(false) closes it, and in-flight Reserve, +// Commit and Release calls are serialized against that swap so a reload can +// never close the store under a lookup. Which store that is — a tenant's share of the // embedded Pebble instance (Embedded.Tenant), a remote backend's view later — // is the opener's business, so every backend gets the same switch semantics. type Managed struct { @@ -44,9 +58,24 @@ func NewManaged(open func() (Deduplicator, error)) *Managed { // already-open store stays open, an already-closed one stays closed. A // failed open leaves the store closed and returns the error — the caller // decides whether that is fatal (boot) or a logged degradation (reload). +// +// A no-op call — the desired state already holds — returns under the read +// lock alone; only a real transition takes the write lock, re-checked once +// held in case another Apply won the race. This matters because a settings +// reload calls Apply for every tenant under the registry lock: on a network +// backend, Commit and Release can hold the read lock for as long as an +// outage lasts, and the write lock waits out every reader, so an +// unconditional write lock here would serialize the whole reload behind +// them, tenant after tenant. func (m *Managed) Apply(enabled bool) error { + if m.settled(enabled) { + return nil + } m.mu.Lock() defer m.mu.Unlock() + if m.settledLocked(enabled) { + return nil + } m.enabled = enabled switch { case enabled && m.db == nil: @@ -63,6 +92,22 @@ func (m *Managed) Apply(enabled bool) error { return nil } +// settled reports whether the store already matches enabled, under its own +// read lock. +func (m *Managed) settled(enabled bool) bool { + m.mu.RLock() + defer m.mu.RUnlock() + return m.settledLocked(enabled) +} + +// settledLocked is settled's condition for a caller already holding mu (read +// or write): an already-open store while enabling, or an already-closed one +// while disabling (db is nil whenever !enabled — Apply's own invariant — so +// disabling never needs the db pointer). +func (m *Managed) settledLocked(enabled bool) bool { + return m.enabled == enabled && (!enabled || m.db != nil) +} + // Open reports whether the store is currently open. func (m *Managed) Open() bool { m.mu.RLock() @@ -70,18 +115,109 @@ func (m *Managed) Open() bool { return m.db != nil } -// CheckAndMark delegates to the open store; ErrDisabled while switched off, +// Reserve collapses a key repeated inside keys to one backend claim — later +// occurrences answer Duplicate — reads a lease <= 0 as DefaultLease, and +// delegates the rest to the open store; ErrDisabled while switched off, // ErrUnavailable while switched on but not open. -func (m *Managed) CheckAndMark(ctx context.Context, eventID string) (bool, error) { +func (m *Managed) Reserve(ctx context.Context, keys []Key, lease time.Duration) ([]Claim, error) { + for _, k := range keys { + if k.Hashed() { + hashedIDCounter.Add(ctx, 1, metric.WithAttributes(attribute.String("table", k.Table))) + } + } + if lease <= 0 { + lease = DefaultLease + } m.mu.RLock() defer m.mu.RUnlock() - if !m.enabled { - return false, ErrDisabled + if err := m.usable(); err != nil { + return nil, err + } + first := make(map[Key]int, len(keys)) + unique := make([]Key, 0, len(keys)) + for _, k := range keys { + if _, seen := first[k]; !seen { + first[k] = len(unique) + unique = append(unique, k) + } } - if m.db == nil { - return false, ErrUnavailable + got, err := m.db.Reserve(ctx, unique, lease) + if err != nil { + return nil, err + } + if len(got) != len(unique) { + if claimed := claimedOnly(got); len(claimed) > 0 { + _ = m.db.Release(context.WithoutCancel(ctx), claimed) + } + return nil, fmt.Errorf("dedupe backend answered %d claims for %d keys", len(got), len(unique)) + } + if len(unique) == len(keys) { + return got, nil + } + claims := make([]Claim, len(keys)) + answered := make([]bool, len(unique)) + for i, k := range keys { + j := first[k] + if answered[j] { + claims[i] = Claim{Key: k, Status: Duplicate} + continue + } + answered[j] = true + claims[i] = got[j] } - return m.db.CheckAndMark(ctx, eventID) + return claims, nil +} + +// Commit delegates the Claimed claims to the open store, with Reserve's +// switch semantics. +func (m *Managed) Commit(ctx context.Context, claims []Claim, retention time.Duration) error { + return m.withClaimed(claims, func(db Deduplicator, claimed []Claim) error { + return db.Commit(ctx, claimed, retention) + }) +} + +// Release delegates the Claimed claims to the open store, with Reserve's +// switch semantics. +func (m *Managed) Release(ctx context.Context, claims []Claim) error { + return m.withClaimed(claims, func(db Deduplicator, claimed []Claim) error { + return db.Release(ctx, claimed) + }) +} + +func (m *Managed) withClaimed(claims []Claim, do func(Deduplicator, []Claim) error) error { + claimed := claimedOnly(claims) + if len(claimed) == 0 { + return nil + } + m.mu.RLock() + defer m.mu.RUnlock() + if err := m.usable(); err != nil { + return err + } + return do(m.db, claimed) +} + +// claimedOnly is the claims a backend's Commit and Release may be handed. +func claimedOnly(claims []Claim) []Claim { + out := make([]Claim, 0, len(claims)) + for _, c := range claims { + if c.Status == Claimed { + out = append(out, c) + } + } + return out +} + +// usable is the switch's answer: nil when the store may be called. Callers +// hold mu. +func (m *Managed) usable() error { + switch { + case !m.enabled: + return ErrDisabled + case m.db == nil: + return ErrUnavailable + } + return nil } // Close releases the store if open. Safe to call when already closed. diff --git a/internal/dedupe/managed_test.go b/internal/dedupe/managed_test.go index 74904be69..cbc4593b6 100644 --- a/internal/dedupe/managed_test.go +++ b/internal/dedupe/managed_test.go @@ -4,6 +4,7 @@ import ( "context" "errors" "testing" + "time" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -16,28 +17,28 @@ func TestManaged_FollowsEnabled(t *testing.T) { ctx := context.Background() assert.False(t, m.Open()) - _, err := m.CheckAndMark(ctx, "e1") + _, err := mark(ctx, m, "e1") require.ErrorIs(t, err, ErrDisabled) require.NoError(t, m.Apply(true)) require.NoError(t, m.Apply(true), "re-applying the same state is a no-op") assert.True(t, m.Open()) - dup, err := m.CheckAndMark(ctx, "e1") + dup, err := mark(ctx, m, "e1") require.NoError(t, err) assert.False(t, dup) - dup, err = m.CheckAndMark(ctx, "e1") + dup, err = mark(ctx, m, "e1") require.NoError(t, err) assert.True(t, dup) require.NoError(t, m.Apply(false)) require.NoError(t, m.Apply(false)) assert.False(t, m.Open()) - _, err = m.CheckAndMark(ctx, "e1") + _, err = mark(ctx, m, "e1") require.ErrorIs(t, err, ErrDisabled) // Re-enabling reopens the same instance: previously seen ids persist. require.NoError(t, m.Apply(true)) - dup, err = m.CheckAndMark(ctx, "e1") + dup, err = mark(ctx, m, "e1") require.NoError(t, err) assert.True(t, dup, "toggling off and on must not forget seen ids") } @@ -45,16 +46,41 @@ func TestManaged_FollowsEnabled(t *testing.T) { // memDedup is the smallest possible backend: what a shared remote store's // per-tenant view would be, minus the network. type memDedup struct { - seen map[string]bool - closed bool + seen map[Key]bool + closed bool + reserved [][]Key // every Reserve's keys, as the backend saw them + leases []time.Duration + released []Claim + short bool // answer one claim too few } -func (m *memDedup) CheckAndMark(_ context.Context, id string) (bool, error) { - if m.seen[id] { - return true, nil +func (m *memDedup) Reserve(_ context.Context, keys []Key, lease time.Duration) ([]Claim, error) { + m.reserved = append(m.reserved, keys) + m.leases = append(m.leases, lease) + claims := make([]Claim, 0, len(keys)) + for _, k := range keys { + st := Claimed + if m.seen[k] { + st = Duplicate + } + claims = append(claims, Claim{Key: k, Status: st, Token: "t"}) } - m.seen[id] = true - return false, nil + if m.short { + claims = claims[1:] + } + return claims, nil +} + +func (m *memDedup) Commit(_ context.Context, claims []Claim, _ time.Duration) error { + for _, c := range claims { + m.seen[c.Key] = true + } + return nil +} + +func (m *memDedup) Release(_ context.Context, claims []Claim) error { + m.released = append(m.released, claims...) + return nil } func (m *memDedup) Close() error { m.closed = true; return nil } @@ -63,16 +89,16 @@ func (m *memDedup) Close() error { m.closed = true; return nil } func TestManaged_AnyBackend(t *testing.T) { t.Parallel() ctx := context.Background() - backend := &memDedup{seen: map[string]bool{}} + backend := &memDedup{seen: map[Key]bool{}} m := NewManaged(func() (Deduplicator, error) { return backend, nil }) - _, err := m.CheckAndMark(ctx, "e1") + _, err := mark(ctx, m, "e1") require.ErrorIs(t, err, ErrDisabled) require.NoError(t, m.Apply(true)) - dup, err := m.CheckAndMark(ctx, "e1") + dup, err := mark(ctx, m, "e1") require.NoError(t, err) assert.False(t, dup) - dup, err = m.CheckAndMark(ctx, "e1") + dup, err = mark(ctx, m, "e1") require.NoError(t, err) assert.True(t, dup) require.NoError(t, m.Close()) @@ -81,7 +107,7 @@ func TestManaged_AnyBackend(t *testing.T) { failing := NewManaged(func() (Deduplicator, error) { return nil, errors.New("backend down") }) require.ErrorContains(t, failing.Apply(true), "backend down") assert.False(t, failing.Open()) - _, err = failing.CheckAndMark(ctx, "e1") + _, err = mark(ctx, failing, "e1") require.ErrorIs(t, err, ErrUnavailable) } @@ -90,9 +116,110 @@ func TestManaged_OpenFailureStaysClosed(t *testing.T) { m := NewManaged(func() (Deduplicator, error) { return nil, errors.New("disk full") }) require.Error(t, m.Apply(true)) assert.False(t, m.Open()) - _, err := m.CheckAndMark(context.Background(), "e1") + _, err := mark(context.Background(), m, "e1") require.ErrorIs(t, err, ErrUnavailable, "switched on but not open must fail closed, not read as disabled") require.NoError(t, m.Close()) - _, err = m.CheckAndMark(context.Background(), "e1") + _, err = mark(context.Background(), m, "e1") require.ErrorIs(t, err, ErrDisabled) } + +// Managed collapses a key repeated in one call before the backend sees it, +// so every backend answers repeats alike, and hands the backend only the +// claims it made. +func TestManaged_CollapsesRepeats(t *testing.T) { + t.Parallel() + ctx := context.Background() + backend := &memDedup{seen: map[Key]bool{}} + m := NewManaged(func() (Deduplicator, error) { return backend, nil }) + require.NoError(t, m.Apply(true)) + a, b := Key{Table: "t", ID: "a"}, Key{Table: "t", ID: "b"} + + claims, err := m.Reserve(ctx, []Key{a, b, a}, time.Second) + require.NoError(t, err) + assert.Equal(t, [][]Key{{a, b}}, backend.reserved, "the backend sees each key once") + assert.Equal(t, []Claim{{Key: a, Status: Claimed, Token: "t"}, {Key: b, Status: Claimed, Token: "t"}, {Key: a, Status: Duplicate}}, claims) + + _, err = m.Reserve(ctx, []Key{{Table: "t", ID: "c"}}, 0) + require.NoError(t, err) + assert.Equal(t, []time.Duration{time.Second, DefaultLease}, backend.leases, "no lease is the default, never an already-lapsed claim") + + backend.short = true + backend.seen[b] = true + _, err = m.Reserve(ctx, []Key{{Table: "t", ID: "d"}, b, {Table: "t", ID: "e"}}, time.Second) + require.ErrorContains(t, err, "answered 2 claims for 3 keys", "a backend answering the wrong count is refused, not indexed past") + assert.Equal(t, []Claim{{Key: Key{Table: "t", ID: "e"}, Status: Claimed, Token: "t"}}, backend.released, + "and gets back only the claims it made, never its Duplicate") +} + +// Commit and Release follow the switch like Reserve, and a call with no +// Claimed claim never reaches the backend. +func TestManaged_CommitAndReleaseFollowTheSwitch(t *testing.T) { + t.Parallel() + ctx := context.Background() + claimed := []Claim{{Key: Key{Table: "t", ID: "a"}, Status: Claimed, Token: "t"}} + m := NewManaged(func() (Deduplicator, error) { return nil, errors.New("down") }) + require.ErrorIs(t, m.Commit(ctx, claimed, 0), ErrDisabled) + require.ErrorIs(t, m.Release(ctx, claimed), ErrDisabled) + require.NoError(t, m.Commit(ctx, []Claim{{Status: Duplicate}}, 0), "nothing to commit") + + require.Error(t, m.Apply(true)) + require.ErrorIs(t, m.Commit(ctx, claimed, 0), ErrUnavailable) + require.ErrorIs(t, m.Release(ctx, claimed), ErrUnavailable) +} + +// blockingDedup's Commit blocks until unblock is closed, standing in for a +// network backend mid-outage: the caller holds Managed's read lock for as +// long as the call takes. +type blockingDedup struct { + memDedup + inCommit chan struct{} // closed once Commit is entered + unblock chan struct{} +} + +func (b *blockingDedup) Commit(ctx context.Context, claims []Claim, retention time.Duration) error { + close(b.inCommit) + <-b.unblock + return b.memDedup.Commit(ctx, claims, retention) +} + +// A no-op Apply must not queue behind an in-flight Commit: it settles under +// the read lock alone, so a reload naming the same state for every tenant +// never waits out another tenant's slow backend call. A real transition is +// the opposite — it still needs the store quiescent, so it waits for Commit +// to finish before touching it. +func TestManaged_ApplyNoOpDoesNotWaitOnCommit(t *testing.T) { + t.Parallel() + backend := &blockingDedup{memDedup: memDedup{seen: map[Key]bool{}}, inCommit: make(chan struct{}), unblock: make(chan struct{})} + m := NewManaged(func() (Deduplicator, error) { return backend, nil }) + require.NoError(t, m.Apply(true)) + + claimed := []Claim{{Key: Key{Table: "t", ID: "a"}, Status: Claimed, Token: "t"}} + commitDone := make(chan error, 1) + go func() { commitDone <- m.Commit(context.Background(), claimed, 0) }() + <-backend.inCommit // Commit is inside the backend call, holding the read lock + + noop := make(chan error, 1) + go func() { noop <- m.Apply(true) }() + select { + case err := <-noop: + require.NoError(t, err) + case <-time.After(2 * time.Second): + t.Fatal("Apply(true) blocked behind an in-flight Commit for a state that already held") + } + + // A real transition is the genuine case: it must wait for Commit, not + // race it — assert it's still pending, then let Commit finish and + // confirm Apply(false) then proceeds and closes the store. + transition := make(chan error, 1) + go func() { transition <- m.Apply(false) }() + select { + case err := <-transition: + t.Fatalf("Apply(false) returned (%v) before the in-flight Commit finished", err) + case <-time.After(50 * time.Millisecond): + } + + close(backend.unblock) + require.NoError(t, <-commitDone) + require.NoError(t, <-transition) + assert.True(t, backend.closed) +} diff --git a/internal/dedupe/stores.go b/internal/dedupe/stores.go index 614912fec..5bb393666 100644 --- a/internal/dedupe/stores.go +++ b/internal/dedupe/stores.go @@ -16,6 +16,25 @@ import ( // nothing that holds the Stores changes with it. type Factory func(id tenant.ID) *Managed +// Gated returns a Factory whose stores open only once ready returns nil, its +// error being the open's: a store switched on meanwhile stays closed and +// fails closed (ErrUnavailable) until an Apply finds the backend ready. For a +// backend whose tenant opens are free but whose shared resource (a remote +// table) is checked once. +func (f Factory) Gated(ready func() error) Factory { + return func(id tenant.ID) *Managed { + m := f(id) + open := m.open + m.open = func() (Deduplicator, error) { + if err := ready(); err != nil { + return nil, err + } + return open() + } + return m + } +} + // Stores is one Managed store per tenant (#583 story 7), each following its // own tenant's dedupe.enabled through Apply. A store is built on first use // and forgotten by Retain once its tenant is no longer served; its seen ids diff --git a/internal/dedupe/stores_test.go b/internal/dedupe/stores_test.go index 15136343e..03e6ecc66 100644 --- a/internal/dedupe/stores_test.go +++ b/internal/dedupe/stores_test.go @@ -2,6 +2,7 @@ package dedupe import ( "context" + "errors" "testing" "time" @@ -30,7 +31,7 @@ func TestStores_ForBuildsOneClosedStorePerTenant(t *testing.T) { assert.Same(t, acme, s.For("acme"), "one store per tenant, however often it is named") assert.NotSame(t, acme, s.For("globex")) assert.False(t, acme.Open(), "built closed: nothing opens until the tenant's switch is applied") - _, err := acme.CheckAndMark(ctx, "e1") + _, err := mark(ctx, acme, "e1") require.ErrorIs(t, err, ErrDisabled, "a store not yet applied answers as a disabled one, the reload-window case") assert.NoDirExists(t, e.Dir()) @@ -49,12 +50,12 @@ func TestStores_TenantsDoNotShareSeenIDs(t *testing.T) { } for _, id := range tenants { - dup, err := s.For(id).CheckAndMark(ctx, "e1") + dup, err := mark(ctx, s.For(id), "e1") require.NoError(t, err) assert.False(t, dup, "%s: the same event id is first seen in each tenant", id) } for _, id := range tenants { - dup, err := s.For(id).CheckAndMark(ctx, "e1") + dup, err := mark(ctx, s.For(id), "e1") require.NoError(t, err) assert.True(t, dup, "%s: and a duplicate within its own tenant", id) } @@ -67,7 +68,7 @@ func TestStores_RetainClosesTheRestAndKeepsTheirData(t *testing.T) { acme, globex := s.For("acme"), s.For("globex") require.NoError(t, acme.Apply(true)) require.NoError(t, globex.Apply(true)) - _, err := acme.CheckAndMark(ctx, "e1") + _, err := mark(ctx, acme, "e1") require.NoError(t, err) require.NoError(t, s.Retain(func(id tenant.ID) bool { return id == "globex" })) @@ -79,7 +80,7 @@ func TestStores_RetainClosesTheRestAndKeepsTheirData(t *testing.T) { restored := s.For("acme") assert.NotSame(t, acme, restored, "the closed store was forgotten") require.NoError(t, restored.Apply(true)) - dup, err := restored.CheckAndMark(ctx, "e1") + dup, err := mark(ctx, restored, "e1") require.NoError(t, err) assert.True(t, dup, "an id seen before the tenant was dropped is still seen") } @@ -88,8 +89,12 @@ func TestStores_RetainClosesTheRestAndKeepsTheirData(t *testing.T) { // slow I/O, as the last Pebble close waiting on a compaction. type gatedDedup struct{ entered, release chan struct{} } -func (g *gatedDedup) CheckAndMark(context.Context, string) (bool, error) { return false, nil } -func (g *gatedDedup) Close() error { g.entered <- struct{}{}; <-g.release; return nil } +func (g *gatedDedup) Reserve(context.Context, []Key, time.Duration) ([]Claim, error) { + return nil, nil +} +func (g *gatedDedup) Commit(context.Context, []Claim, time.Duration) error { return nil } +func (g *gatedDedup) Release(context.Context, []Claim) error { return nil } +func (g *gatedDedup) Close() error { g.entered <- struct{}{}; <-g.release; return nil } // One tenant's I/O is that tenant's wait alone: Retain edits the map under // the lock and closes outside it, so a dropped tenant's slow close never @@ -130,3 +135,21 @@ func TestStores_CloseClosesEveryStore(t *testing.T) { assert.False(t, e.Open(), "the instance closes with the last store") require.NoError(t, s.Close(), "closing again is a no-op") } + +func TestFactory_GatedOpensOnlyOnceReady(t *testing.T) { + t.Parallel() + notReady := errors.New("table missing") + ready := notReady + gated := NewStores(Factory(NewEmbedded(t.TempDir()).Tenant).Gated(func() error { return ready })) + t.Cleanup(func() { _ = gated.Close() }) + acme := gated.For("acme") + + require.ErrorIs(t, acme.Apply(true), notReady) + assert.False(t, acme.Open()) + _, err := mark(context.Background(), acme, "e1") + require.ErrorIs(t, err, ErrUnavailable, "switched on but not ready: fails closed, never open") + + ready = nil + require.NoError(t, acme.Apply(true), "the next apply finds it ready") + assert.True(t, acme.Open()) +} diff --git a/internal/dedupe/sweep.go b/internal/dedupe/sweep.go new file mode 100644 index 000000000..25759f39b --- /dev/null +++ b/internal/dedupe/sweep.go @@ -0,0 +1,210 @@ +package dedupe + +import ( + "bytes" + "context" + "errors" + "fmt" + "log/slog" + "time" + + "github.com/cockroachdb/pebble" + "go.opentelemetry.io/otel" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/metric" +) + +// The sweep's cadence. Expired keys are already absent to Reserve, so the +// sweep only reclaims space and can run rarely; the first pass comes soon +// after the instance opens so an upgrade's version-0 keys go without waiting +// an hour. +const ( + sweepInterval = time.Hour + sweepFirstDelay = time.Minute + // sweepChunk keys are read per chunk, with sweepPause between chunks: at + // most ~100k keys a second. A Commit waits only for a chunk's re-reads + // and deletes, never for its read, however many tombstones it skips. + sweepChunk = 1024 + sweepPause = 10 * time.Millisecond +) + +// Swept-key reasons, the metric's reason attribute. +const ( + sweptExpired = "expired" + sweptVersion0 = "version_0" + sweptAttribute = "reason" +) + +var sweptKeysCounter, _ = otel.Meter("wavehouse-dedupe").Int64Counter( + "wavehouse_dedupe_swept_keys_total", + metric.WithDescription("Keys the embedded dedupe sweep deleted, by reason: expired (retention ended) or version_0 (the layout before ids were keyed by table)"), +) + +// sweepResult is what a sweep deleted. +type sweepResult struct { + Expired, Version0 int +} + +// startSweep runs the sweep over db until the returned stop is called; stop +// waits for a chunk in progress to finish. Callers hold e.mu. +func (e *Embedded) startSweep(db *pebble.DB) (stop func()) { + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { + defer close(done) + wait := e.sweepFirst + for { + select { + case <-ctx.Done(): + return + case <-time.After(wait): + } + wait = e.sweepEvery + res, err := e.sweep(ctx, db) + switch { + case err != nil && ctx.Err() == nil: + slog.WarnContext(ctx, "dedupe sweep failed; retrying next interval", "error", err, "expired", res.Expired, "version_0", res.Version0) + case res.Expired+res.Version0 > 0: + slog.InfoContext(ctx, "dedupe sweep deleted keys", "expired", res.Expired, "version_0", res.Version0) + } + } + }() + return func() { + cancel() + <-done + } +} + +// sweep makes one pass over the whole instance, deleting keys whose +// retention has ended and version-0 keys, which never count: those from +// before ids were keyed by table (tenant ‖ 0x00 ‖ id, or the bare id before +// that). They are told apart by value, since a bare id may be any bytes, a +// current key's included: only commits are stored, and every commit has the +// committedMark layout. It stops early, without error, when ctx ends. +func (e *Embedded) sweep(ctx context.Context, db *pebble.DB) (sweepResult, error) { + var res sweepResult + var from []byte + for { + next, err := e.sweepChunk(ctx, db, from, &res) + if err != nil || next == nil { + return res, err + } + from = next + select { + case <-ctx.Done(): + return res, nil + case <-time.After(sweepPause): + } + } +} + +// sweepChunk deletes the sweepable keys among the next sweepChunk keys from +// from, returning where the next chunk starts (nil at the end). It reads them +// without commitMu, since Pebble skips the tombstones between keys inside the +// read and a run of them left by an earlier pass would otherwise hold every +// Commit for its whole length. +func (e *Embedded) sweepChunk(ctx context.Context, db *pebble.DB, from []byte, res *sweepResult) ([]byte, error) { + candidates, next, err := sweepCandidates(db, from, e.now(), e.sweepReadHook) + if err != nil || len(candidates) == 0 { + return next, err + } + if e.sweepScanHook != nil { + e.sweepScanHook() + } + expired, v0, err := e.deleteSweepable(db, candidates) + if err != nil { + return nil, err + } + res.Expired += int(expired) + res.Version0 += int(v0) + if expired > 0 { + sweptKeysCounter.Add(ctx, expired, metric.WithAttributes(attribute.String(sweptAttribute, sweptExpired))) + } + if v0 > 0 { + sweptKeysCounter.Add(ctx, v0, metric.WithAttributes(attribute.String(sweptAttribute, sweptVersion0))) + } + return next, nil +} + +// sweepCandidates reads the next sweepChunk keys, starting at from, and +// returns those sweepable at now and where the next chunk starts (nil at the +// end). onKey, when non-nil, runs once per key visited, before it is +// evaluated — a test hook proving this read holds no lock while it runs. +func sweepCandidates(db *pebble.DB, from []byte, now time.Time, onKey func()) (candidates [][]byte, next []byte, err error) { + it, err := db.NewIter(&pebble.IterOptions{LowerBound: from}) + if err != nil { + return nil, nil, fmt.Errorf("dedupe sweep: %w", err) + } + seen := 0 + for valid := it.First(); valid; valid = it.Next() { + if onKey != nil { + onKey() + } + if seen == sweepChunk { + next = bytes.Clone(it.Key()) + break + } + seen++ + if sweepReason(it.Value(), now) != "" { + candidates = append(candidates, bytes.Clone(it.Key())) + } + } + if err := it.Close(); err != nil { + return nil, nil, fmt.Errorf("dedupe sweep: %w", err) + } + return candidates, next, nil +} + +// deleteSweepable re-reads each candidate and deletes those still sweepable, +// holding commitMu so no Commit lands between the re-read and the delete: a +// key re-committed after the unlocked read is never deleted with its new +// value. +func (e *Embedded) deleteSweepable(db *pebble.DB, candidates [][]byte) (expired, v0 int64, err error) { + e.commitMu.Lock() + defer e.commitMu.Unlock() + now := e.now() + b := db.NewBatch() + defer func() { _ = b.Close() }() + for _, k := range candidates { + val, closer, err := db.Get(k) + if errors.Is(err, pebble.ErrNotFound) { + continue + } + if err != nil { + return 0, 0, fmt.Errorf("dedupe sweep: %w", err) + } + reason := sweepReason(val, now) + _ = closer.Close() + switch reason { + case sweptVersion0: + v0++ + case sweptExpired: + expired++ + default: + continue + } + if err := b.Delete(k, nil); err != nil { + return 0, 0, fmt.Errorf("dedupe sweep: %w", err) + } + } + if e.sweepDeleteHook != nil { + e.sweepDeleteHook() + } + // NoSync: a delete lost to a crash is redone by the next pass. + if err := b.Commit(pebble.NoSync); err != nil { + return 0, 0, fmt.Errorf("dedupe sweep: %w", err) + } + return expired, v0, nil +} + +// sweepReason is why the sweep deletes a key holding val at now, or "" when +// it keeps it. +func sweepReason(val []byte, now time.Time) string { + switch { + case !isCommit(val): + return sweptVersion0 + case committedExpired(val, now): + return sweptExpired + } + return "" +} diff --git a/internal/dedupe/sweep_test.go b/internal/dedupe/sweep_test.go new file mode 100644 index 000000000..ee87088e2 --- /dev/null +++ b/internal/dedupe/sweep_test.go @@ -0,0 +1,253 @@ +package dedupe + +import ( + "context" + "errors" + "fmt" + "math" + "sync/atomic" + "testing" + "time" + + "github.com/cockroachdb/pebble" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// stepClock is a clock a test moves by hand. +type stepClock struct{ ns atomic.Int64 } + +func newStepClock() *stepClock { + c := &stepClock{} + c.ns.Store(time.Now().UnixNano()) + return c +} + +func (c *stepClock) now() time.Time { return time.Unix(0, c.ns.Load()) } +func (c *stepClock) advance(d time.Duration) { c.ns.Add(int64(d)) } +func present(t *testing.T, e *Embedded, key []byte) bool { + t.Helper() + _, closer, err := e.db.Get(key) + if errors.Is(err, pebble.ErrNotFound) { + return false + } + require.NoError(t, err) + _ = closer.Close() + return true +} + +// commitIDs reserves and commits ids in table "events" with retention. +func commitIDs(t *testing.T, m *Managed, retention time.Duration, ids ...string) { + t.Helper() + keys := make([]Key, len(ids)) + for i, id := range ids { + keys[i] = Key{Table: "events", ID: id} + } + claims, err := m.Reserve(context.Background(), keys, DefaultLease) + require.NoError(t, err) + require.NoError(t, m.Commit(context.Background(), claims, retention)) +} + +// A sweep deletes the keys whose retention has ended and every version-0 +// key, across chunk boundaries, and leaves every live key: one kept forever, +// one not yet expired, and one that expired and was committed again. +func TestEmbedded_SweepDeletesExpiredAndVersionZeroKeys(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + clock := newStepClock() + SetClock(e, clock.now) + acme, globex := switchedOn(t, e, "acme"), switchedOn(t, e, "globex") + + // More expired keys than one chunk holds, interleaved with live ones. + var expired, live []string + for i := range 2*sweepChunk + 10 { + expired = append(expired, fmt.Sprintf("x%05d", i)) + live = append(live, fmt.Sprintf("x%05d-live", i)) + } + commitIDs(t, acme, time.Hour, expired...) + commitIDs(t, acme, 3*time.Hour, live...) + commitIDs(t, globex, 0, "forever") + commitIDs(t, globex, time.Hour, "recommitted") + // A bare id from before tenants led the key may spell a current key; + // its value tells it apart. + for _, k := range []string{"acme\x00e1", "acme\x00e2", "globex\x00e1", "acme/events/stale"} { + require.NoError(t, e.db.Set([]byte(k), make([]byte, 8), pebble.Sync)) + } + + clock.advance(2 * time.Hour) + commitIDs(t, globex, time.Hour, "recommitted") + res, err := e.sweep(context.Background(), e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{Expired: len(expired), Version0: 4}, res) + + for _, id := range expired { + require.False(t, present(t, e, AppendKey(nil, KeyPrefix("acme"), Key{Table: "events", ID: id})), id) + } + for _, id := range live { + require.True(t, present(t, e, AppendKey(nil, KeyPrefix("acme"), Key{Table: "events", ID: id})), id) + } + assert.True(t, present(t, e, AppendKey(nil, KeyPrefix("globex"), Key{Table: "events", ID: "forever"}))) + assert.True(t, present(t, e, AppendKey(nil, KeyPrefix("globex"), Key{Table: "events", ID: "recommitted"}))) + assert.False(t, present(t, e, []byte("acme\x00e1"))) + assert.False(t, present(t, e, []byte("globex\x00e1"))) + assert.False(t, present(t, e, []byte("acme/events/stale"))) + + dup, err := mark(context.Background(), globex, "recommitted") + require.NoError(t, err) + assert.True(t, dup, "the new commit survived the sweep") + res, err = e.sweep(context.Background(), e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{}, res, "a second pass finds nothing") +} + +// expiredAndClaimed commits id "e1" with an hour's retention, lets it expire +// and claims it again, for a test to commit mid-sweep. +func expiredAndClaimed(t *testing.T) (*Embedded, *Managed, []Claim) { + t.Helper() + e := NewEmbedded(t.TempDir()) + clock := newStepClock() + SetClock(e, clock.now) + m := switchedOn(t, e, "acme") + commitIDs(t, m, time.Hour, "e1") + clock.advance(2 * time.Hour) + claims, err := m.Reserve(context.Background(), []Key{{Table: "events", ID: "e1"}}, DefaultLease) + require.NoError(t, err) + require.Equal(t, Claimed, claims[0].Status, "expired: claimable again") + return e, m, claims +} + +// A key committed again after a sweep chunk read it as expired, but before +// the chunk re-read it, is kept: the re-read sees the new commit. +func TestEmbedded_SweepKeepsAKeyCommittedAfterItsRead(t *testing.T) { + t.Parallel() + e, m, claims := expiredAndClaimed(t) + var commitErr error + e.sweepScanHook = func() { commitErr = m.Commit(context.Background(), claims, time.Hour) } + res, err := e.sweep(context.Background(), e.db) + require.NoError(t, err) + require.NoError(t, commitErr) + assert.Equal(t, sweepResult{}, res) + + dup, err := mark(context.Background(), m, "e1") + require.NoError(t, err) + assert.True(t, dup, "the commit made after the read survived the sweep") +} + +// A Commit that arrives while a sweep chunk has re-read an expired key but +// not yet deleted it waits for the chunk, so the new commit is never deleted +// with the old value. Without the lock the Commit lands in the gap and the +// sweep then deletes it; the wait below only ever lets that pass, never fail. +func TestEmbedded_SweepNeverDeletesACommitLandingMidChunk(t *testing.T) { + t.Parallel() + e, m, claims := expiredAndClaimed(t) + done := make(chan error, 1) + e.sweepDeleteHook = func() { + go func() { done <- m.Commit(context.Background(), claims, time.Hour) }() + select { + case err := <-done: + done <- err + case <-time.After(50 * time.Millisecond): + } + } + res, err := e.sweep(context.Background(), e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{Expired: 1}, res) + require.NoError(t, <-done) + + dup, err := mark(context.Background(), m, "e1") + require.NoError(t, err) + assert.True(t, dup, "the commit made mid-chunk survived the sweep") +} + +// sweepCandidates' read never holds commitMu, over a fixture with a few +// tombstones ahead of the one live key it finds sweepable. +func TestEmbedded_SweepReadRunsUnlocked(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + switchedOn(t, e, "acme") + b := e.db.NewBatch() + for i := range 4 { + require.NoError(t, b.Delete(fmt.Appendf(nil, "acme\x00%02d", i), nil)) + } + require.NoError(t, b.Set([]byte("acme\x01"), make([]byte, 8), nil)) + require.NoError(t, b.Commit(pebble.NoSync)) + require.NoError(t, e.db.Flush()) + + var visits int + var sawLocked bool + e.sweepReadHook = func() { + visits++ + // Non-blocking, on the reading goroutine: fails if the read holds commitMu. + if e.commitMu.TryLock() { + e.commitMu.Unlock() + } else { + sawLocked = true + } + } + + res, err := e.sweep(context.Background(), e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{Version0: 1}, res) + assert.Positive(t, visits, "the read hook ran") + assert.False(t, sawLocked, "commitMu must be free while sweepCandidates' read is running") +} + +// A retention is honoured on read before any sweep has run: the key is a +// duplicate until the retention ends and claimable from that instant. +func TestEmbedded_RetentionHonouredOnRead(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + clock := newStepClock() + SetClock(e, clock.now) + m := switchedOn(t, e, "acme") + commitIDs(t, m, time.Hour, "e1") + + clock.advance(time.Hour - time.Nanosecond) + claims, err := m.Reserve(context.Background(), []Key{{Table: "events", ID: "e1"}}, DefaultLease) + require.NoError(t, err) + assert.Equal(t, Duplicate, claims[0].Status) + + clock.advance(time.Nanosecond) + claims, err = m.Reserve(context.Background(), []Key{{Table: "events", ID: "e1"}}, DefaultLease) + require.NoError(t, err) + assert.Equal(t, Claimed, claims[0].Status) +} + +// The sweep runs on its own once the instance opens, and stops with it. +func TestEmbedded_SweepRunsWhileOpen(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + e.sweepFirst, e.sweepEvery = time.Millisecond, time.Millisecond + m := e.Tenant("acme") + require.NoError(t, m.Apply(true)) + require.NoError(t, e.db.Set([]byte("acme\x00e1"), make([]byte, 8), pebble.Sync)) + assert.Eventually(t, func() bool { return !present(t, e, []byte("acme\x00e1")) }, 5*time.Second, 5*time.Millisecond) + require.NoError(t, m.Apply(false), "closing waits for the sweep to stop") + assert.False(t, e.Open()) +} + +// A sweep stops between chunks when its context ends. +func TestEmbedded_SweepStopsWhenCancelled(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + switchedOn(t, e, "acme") + b := e.db.NewBatch() + for i := range 3 * sweepChunk { + require.NoError(t, b.Set(fmt.Appendf(nil, "acme\x00%05d", i), nil, nil)) + } + require.NoError(t, b.Commit(pebble.Sync)) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + res, err := e.sweep(ctx, e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{Version0: sweepChunk}, res, "one chunk, then the cancellation is seen") +} + +func TestExpiry(t *testing.T) { + t.Parallel() + now := time.Unix(0, 1_000) + assert.Zero(t, expiry(now, 0)) + assert.Zero(t, expiry(now, -time.Second)) + assert.Equal(t, 1_000+int64(time.Hour), expiry(now, time.Hour)) + assert.Equal(t, int64(math.MaxInt64), expiry(now, time.Duration(math.MaxInt64)), "saturates rather than wrapping into the past") +} diff --git a/internal/ingest/worker_test.go b/internal/ingest/worker_test.go index 354292ab7..42251f0da 100644 --- a/internal/ingest/worker_test.go +++ b/internal/ingest/worker_test.go @@ -254,10 +254,7 @@ func TestStartIngestWorker_StopFunc_RespectsShutdownDeadline(t *testing.T) { <-release w.WriteHeader(http.StatusOK) })) - t.Cleanup(func() { - close(release) - chSrv.Close() - }) + t.Cleanup(chSrv.Close) u, _ := url.Parse(chSrv.URL) host, port, _ := net.SplitHostPort(u.Host) @@ -269,6 +266,12 @@ func TestStartIngestWorker_StopFunc_RespectsShutdownDeadline(t *testing.T) { return chconn.Target{URL: fmt.Sprintf("http://%s:%s", host, port), Username: "u", Password: "p", Database: "db"} }, nil) require.NoError(t, err) + // The deadline below abandons the worker, not its insert: finish the insert + // and join the worker before the broker closes under its ack. + t.Cleanup(func() { + close(release) + assert.NoError(t, stopFn(context.Background())) + }) // Publish so there's an in-flight insert blocking on `release`. err = emb.Publish(ctx, mq.Topic{Tenant: tenant.Default, Table: "events"}, makeEnvelope(t, "events", "", map[string]any{"id": 1})) diff --git a/internal/keyenc/keyenc.go b/internal/keyenc/keyenc.go index 46e1f6f0a..476e0f27f 100644 --- a/internal/keyenc/keyenc.go +++ b/internal/keyenc/keyenc.go @@ -1,14 +1,15 @@ // Package keyenc is the one escaping composite WaveHouse keys are built -// from: NATS subject tokens and cache keys. A field keeps ASCII letters, -// digits, '_' and '-' as they are and writes every other byte as %XX +// from: NATS subject tokens, cache keys and dedupe keys. A field keeps ASCII +// letters, digits, '_' and '-' as they are and writes every other byte as %XX // (uppercase hex), so no separator, wildcard, whitespace, brace or non-ASCII // byte ever appears in it unescaped, and any table name ClickHouse accepts // encodes. The bytes it keeps are exactly a tenant id's (tenant.Parse), so a // tenant id is its own escaped form. // -// Keys built from it are stored — queued under NATS subjects, held in caches -// — so a change to what it keeps orphans them. Earlier builds escaped '-' as -// %2D; Unescape still reads that form. +// Keys built from it are stored — queued under NATS subjects, held in caches, +// kept as dedupe keys — so a change to what it keeps orphans them; an +// orphaned dedupe key lets a seen id through again. Earlier builds escaped +// '-' as %2D; Unescape still reads that form. package keyenc import ( diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index 15ca485fe..f852a4c2d 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -227,6 +227,13 @@ func (e *EmbeddedNATS) takeStock(ctx context.Context) error { held uint64 } dlqs := map[tenant.ID]dlqState{} + // duplicates is the ingest stream's own Duplicates window as found on + // disk, keyed alongside dlqs: a stream from before EmbeddedDuplicateWindow + // existed, or reopened under a different value, must not be counted as + // already at budget below, or SetMaxBytes(same budget) short-circuits and + // the stale window is never brought forward (measured: a stream with + // Duplicates=10s kept 10s after NewEmbedded + SetMaxBytes(same budget)). + duplicates := map[tenant.ID]time.Duration{} streams := e.js.ListStreams(ctx) for info := range streams.Info() { name := info.Config.Name @@ -234,6 +241,7 @@ func (e *EmbeddedNATS) takeStock(ctx context.Context) error { q := e.queue(id) q.ingest = true q.asked, q.ingestCap = info.Config.MaxBytes, info.Config.MaxBytes + duplicates[id] = info.Config.Duplicates } else if id, ok := streamTenant(dlqStreamPrefix, name); ok { e.queue(id).dlq = true dlqs[id] = dlqState{limit: info.Config.MaxBytes, held: info.State.Bytes} @@ -244,14 +252,16 @@ func (e *EmbeddedNATS) takeStock(ctx context.Context) error { } // A pair is at its budget when its dead-letter stream is at a tenth of // the ingest cap, or above it holding more than that: the shrink guard's - // doing. Anything else is a pair a stop or a failed update left split, or - // one missing its dead-letter stream, so its budget stays unapplied and - // the boot's SetMaxBytes applies it to both streams again. + // doing, AND its ingest stream's duplicate window already matches + // EmbeddedDuplicateWindow. Anything else is a pair a stop or a failed + // update left split, one missing its dead-letter stream, or one whose + // duplicate window is stale, so its budget stays unapplied and the boot's + // SetMaxBytes applies it — and the current window — to both streams again. for id, q := range e.queues { d, ok := dlqs[id] tenth := q.asked / dlqShare guarded := d.limit > tenth && d.held <= math.MaxInt64 && int64(d.held) > tenth - if q.ingest && ok && (d.limit == tenth || guarded) { + if q.ingest && ok && (d.limit == tenth || guarded) && duplicates[id] == EmbeddedDuplicateWindow { q.maxBytes = q.asked } e.record(id, q) @@ -317,17 +327,29 @@ func (e *EmbeddedNATS) record(id tenant.ID, q *tenantQueue) { } } +// EmbeddedDuplicateWindow is how long an ingest queue remembers a +// WithIdempotencyKey key. It must be at least lease + ceil(lease) + 1s +// (2*lease + 1s for a whole-second lease), which config checks against +// dedupe.lease at boot. A claim left to lapse after an uncertain publish is +// republished once the lease ends, but the in-flight 503 tells a client to +// retry only after the FULL lease, so an obedient client's retry can land up +// to ~2*lease after the original Reserve; the +1s covers a backend (DynamoDB, for one) that rounds a +// claim's expiry up by as much. Only a window at least that long guarantees +// this queue still drops the retry's second copy. +const EmbeddedDuplicateWindow = 2 * time.Minute + // ingestStreamConfig is tenant id's ingest stream. LimitsPolicy: standard // append-only log; the Active Sweeper handles message purging. MaxBytes caps // the tenant's share of the disk. DiscardNew rejects new messages when full, // propagating backpressure to the upstream API — for this tenant alone. func ingestStreamConfig(id tenant.ID, maxBytes int64) jetstream.StreamConfig { return jetstream.StreamConfig{ - Name: ingestStreamName(id), - Subjects: []string{tenantSubjects(ingestPrefix, id)}, - Retention: jetstream.LimitsPolicy, - MaxBytes: maxBytes, - Discard: jetstream.DiscardNew, + Name: ingestStreamName(id), + Subjects: []string{tenantSubjects(ingestPrefix, id)}, + Retention: jetstream.LimitsPolicy, + MaxBytes: maxBytes, + Discard: jetstream.DiscardNew, + Duplicates: EmbeddedDuplicateWindow, } } @@ -1086,7 +1108,10 @@ func (e *EmbeddedNATS) Close() error { // Owning the lifecycle (NoSigs, #287) means waiting it out: without this, // run()'s remaining defers unwind while JetStream is still tearing down // and the process can exit mid-shutdown (as-if-crashed stream state). - // Milliseconds for an in-process server. + // Milliseconds for an in-process server. It does not join a durable's + // state flusher: a write under way can land after Close returns, or never + // if the process exits first, leaving the durable's previous ack state on + // disk (#665). e.server.WaitForShutdown() return nil } diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index 55e7fec42..e1d28281e 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -10,6 +10,7 @@ import ( "time" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" "github.com/nats-io/nats.go" "github.com/nats-io/nats.go/jetstream" "github.com/stretchr/testify/assert" @@ -19,26 +20,6 @@ import ( // testBudget is the byte budget newTestEmbedded opens each queue at. const testBudget = 64 << 20 -// storeDir is a temporary directory for a broker's store whose removal -// retries briefly: a consumer's state file can land after Close has returned, -// which fails t.TempDir's one-shot RemoveAll (#442). The retrying cleanup runs -// first (cleanups are LIFO), leaving t.TempDir an empty directory to remove. -func storeDir(t *testing.T) string { - t.Helper() - dir := filepath.Join(t.TempDir(), "store") - t.Cleanup(func() { - var err error - for range 50 { - if err = os.RemoveAll(dir); err == nil { - return - } - time.Sleep(20 * time.Millisecond) - } - t.Errorf("remove %s: %v", dir, err) - }) - return dir -} - // openEmbedded starts an EmbeddedNATS over dir, closed by the test framework. func openEmbedded(t *testing.T, dir string) *EmbeddedNATS { t.Helper() @@ -53,7 +34,7 @@ func openEmbedded(t *testing.T, dir string) *EmbeddedNATS { // at testBudget. func newTestEmbedded(t *testing.T, tenants ...tenant.ID) *EmbeddedNATS { t.Helper() - e := openEmbedded(t, storeDir(t)) + e := openEmbedded(t, storedir.New(t)) if len(tenants) == 0 { tenants = []tenant.ID{tenant.Default} } @@ -162,12 +143,46 @@ func TestEmbeddedNATS_PublishHeaders(t *testing.T) { assert.Equal(t, []byte("x"), raw.Data) } +// A repeated idempotency key inside the duplicate window is dropped as a +// success, so an uncertain publish can be republished safely. This stream is +// created directly, never recorded by takeStock, so SetMaxBytes's next +// budget apply always runs and picks up the current window; +// TestNewEmbedded_TakeStockRefreshesAStaleDuplicateWindow covers the boot +// path, where takeStock itself must not mistake a stale window for one +// already at budget. +func TestEmbeddedNATS_Publish_IdempotencyKeyDropsARepeat(t *testing.T) { + e := openEmbedded(t, storedir.New(t)) + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + // Explicit rather than the server's default, which happens to match today. + require.Equal(t, EmbeddedDuplicateWindow, ingestStreamConfig(tenant.Default, testBudget).Duplicates) + old := ingestStreamConfig(tenant.Default, testBudget) + old.Duplicates = 10 * time.Second + _, err := e.js.CreateStream(ctx, old) + require.NoError(t, err) + require.NoError(t, e.SetMaxBytes(ctx, tenant.Default, testBudget)) + require.Equal(t, EmbeddedDuplicateWindow, streamConfig(t, e, "INGEST_0").Duplicates) + + topic := Topic{Tenant: tenant.Default, Table: "t"} + require.NoError(t, e.Publish(ctx, topic, []byte("a"), WithIdempotencyKey("k1"))) + require.NoError(t, e.Publish(ctx, topic, []byte("a again"), WithIdempotencyKey("k1")), "a repeat is a success") + require.NoError(t, e.Publish(ctx, topic, []byte("b"), WithIdempotencyKey("k2"))) + require.NoError(t, e.Publish(ctx, topic, []byte("c"))) + + var got []string + require.NoError(t, e.ReplaySince(ctx, topic, time.Time{}, func(data []byte) bool { + got = append(got, string(data)) + return true + })) + assert.Equal(t, []string{"a", "b", "c"}, got) +} + // A tenant's first budget opens its queue: an ingest stream holding its // subjects alone at the budget, refusing when full, and a dead-letter stream // at a tenth of it, dropping its oldest when full. No other tenant gets one. func TestEmbeddedNATS_SetMaxBytes_OpensTheTenantsQueue(t *testing.T) { t.Parallel() - e := openEmbedded(t, storeDir(t)) + e := openEmbedded(t, storedir.New(t)) assert.Zero(t, e.MaxBytes("acme"), "no budget applied yet") require.NoError(t, e.SetMaxBytes(t.Context(), "acme", testBudget)) @@ -177,6 +192,7 @@ func TestEmbeddedNATS_SetMaxBytes_OpensTheTenantsQueue(t *testing.T) { assert.Equal(t, []string{"ingest.acme.>"}, ingest.Subjects) assert.Equal(t, int64(testBudget), ingest.MaxBytes) assert.Equal(t, jetstream.DiscardNew, ingest.Discard) + assert.Equal(t, EmbeddedDuplicateWindow, ingest.Duplicates) dlq := streamConfig(t, e, "DLQ_acme") assert.Equal(t, []string{"dlq.acme.>"}, dlq.Subjects) assert.Equal(t, int64(testBudget)/10, dlq.MaxBytes) @@ -342,7 +358,7 @@ func TestEmbeddedNATS_DefaultLogger(t *testing.T) { t.Parallel() // NewEmbedded without a logger should not panic — it falls back to the // default slog logger. - e, err := NewEmbedded(storeDir(t)) + e, err := NewEmbedded(storedir.New(t)) require.NoError(t, err) t.Cleanup(func() { _ = e.Close() }) } @@ -458,7 +474,7 @@ func TestEmbeddedNATS_SetMaxBytes_IngestFailureChangesNothing(t *testing.T) { // however recently a publish tried. func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { t.Parallel() - dir := storeDir(t) + dir := storedir.New(t) // The dead-letter stream is the first of the pair to open. A failed open // removes what was in the way, so the obstacle is put back before each // attempt meant to fail. @@ -509,7 +525,7 @@ func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { // resize and reload takes. Once the window has passed, a publish tries again. func TestEmbeddedNATS_PacesTheRetriesOfAQueueThatCannotOpen(t *testing.T) { t.Parallel() - dir := storeDir(t) + dir := storedir.New(t) block := filepath.Join(dir, "jetstream", "$G", "streams", dlqStreamName("acme")) obstruct := func() { t.Helper() @@ -574,7 +590,7 @@ func TestEmbeddedNATS_PacesTheRetriesOfAQueueThatCannotOpen(t *testing.T) { // joined, so its row reaches them rather than a stream nobody reads. func TestEmbeddedNATS_Publish_OpensAQueueItsOpenGaveUpOn(t *testing.T) { t.Parallel() - dir := storeDir(t) + dir := storedir.New(t) block := filepath.Join(dir, "jetstream", "$G", "streams", ingestStreamName("acme")) require.NoError(t, os.MkdirAll(filepath.Dir(block), 0o750)) require.NoError(t, os.WriteFile(block, nil, 0o600)) @@ -618,7 +634,7 @@ func TestEmbeddedNATS_SetMaxBytes_UndoRestoresTheIngestStreamsCap(t *testing.T) t.Parallel() ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() - dir := storeDir(t) + dir := storedir.New(t) first, err := NewEmbedded(dir) require.NoError(t, err) require.NoError(t, first.SetMaxBytes(ctx, "acme", 8<<20)) @@ -641,7 +657,7 @@ func TestEmbeddedNATS_SetMaxBytes_UndoRestoresTheIngestStreamsCap(t *testing.T) // queue itself is open, so SetMaxBytes succeeds. func TestEmbeddedNATS_Consume_ReportsAQueueItCannotJoin(t *testing.T) { t.Parallel() - e := openEmbedded(t, storeDir(t)) + e := openEmbedded(t, storedir.New(t)) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() // A durable name the client refuses: with no queue yet, nothing checks it. @@ -666,7 +682,7 @@ func TestEmbeddedNATS_Consume_ReportsAQueueItCannotJoin(t *testing.T) { // stream keeps what it holds, capped at that, and every row survives. func TestEmbeddedNATS_SetMaxBytes_NeverShrinksTheDeadLetterQueueBelowWhatItHolds(t *testing.T) { t.Parallel() - e := openEmbedded(t, storeDir(t)) + e := openEmbedded(t, storedir.New(t)) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() require.NoError(t, e.SetMaxBytes(ctx, "acme", 10<<20)) @@ -881,7 +897,7 @@ func TestEmbeddedNATS_DeadLetter_ReopensAMissingQueue(t *testing.T) { func TestEmbeddedNATS_Publish_QueueFull(t *testing.T) { t.Parallel() - e := openEmbedded(t, storeDir(t)) + e := openEmbedded(t, storedir.New(t)) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() require.NoError(t, e.SetMaxBytes(ctx, "acme", 4<<10)) @@ -1364,7 +1380,7 @@ func TestNewEmbedded_AStoreItCannotCreateFailsAtOnce(t *testing.T) { // so no tenant's queue could open beside them. func TestNewEmbedded_DeletesTheStreamsAnEarlierBuildShared(t *testing.T) { t.Parallel() - dir := storeDir(t) + dir := storedir.New(t) old, err := NewEmbedded(dir) require.NoError(t, err) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) @@ -1395,7 +1411,7 @@ func TestNewEmbedded_ASplitPairIsAppliedAgainAtBoot(t *testing.T) { t.Parallel() ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() - dir := storeDir(t) + dir := storedir.New(t) first, err := NewEmbedded(dir) require.NoError(t, err) for _, id := range []tenant.ID{"split", "gone", "guarded"} { @@ -1433,7 +1449,7 @@ func TestNewEmbedded_ASplitPairIsAppliedAgainAtBoot(t *testing.T) { // so what such a tenant had queued still reaches the worker. func TestNewEmbedded_TakesStockOfTheQueuesOnDisk(t *testing.T) { t.Parallel() - dir := storeDir(t) + dir := storedir.New(t) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() first, err := NewEmbedded(dir) @@ -1469,6 +1485,34 @@ func TestNewEmbedded_TakesStockOfTheQueuesOnDisk(t *testing.T) { assert.Equal(t, int64(8<<20), streamConfig(t, e, "INGEST_acme").MaxBytes) } +// takeStock must not count a stream as at its budget when its Duplicates +// window is stale (from before EmbeddedDuplicateWindow existed, or changed +// underneath it): otherwise SetMaxBytes's same-budget early return never lets +// a later apply bring the window forward, and the stream keeps whatever it +// had indefinitely. +func TestNewEmbedded_TakeStockRefreshesAStaleDuplicateWindow(t *testing.T) { + t.Parallel() + dir := storedir.New(t) + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + first, err := NewEmbedded(dir) + require.NoError(t, err) + require.NoError(t, first.SetMaxBytes(ctx, "acme", 8<<20)) + stale := ingestStreamConfig("acme", 8<<20) + stale.Duplicates = 10 * time.Second + _, err = first.js.UpdateStream(ctx, stale) + require.NoError(t, err) + require.NoError(t, first.Close()) + + e := openEmbedded(t, dir) + require.Equal(t, 10*time.Second, streamConfig(t, e, "INGEST_acme").Duplicates, "the stale window is still on disk") + + require.NoError(t, e.SetMaxBytes(ctx, "acme", 8<<20), "same budget as before") + assert.Equal(t, EmbeddedDuplicateWindow, streamConfig(t, e, "INGEST_acme").Duplicates, + "takeStock must not have marked this pair already at budget, or this apply would have no-op'd") +} + // A durable found on disk is kept as it stands when it holds the settings // asked for — a boot over many queues writes nothing it need not — and is // updated in place when they differ; either way delivery resumes past what it @@ -1484,7 +1528,7 @@ func TestEmbeddedNATS_ADurableOnDiskIsReusedAcrossARestart(t *testing.T) { } { t.Run(tt.name, func(t *testing.T) { t.Parallel() - dir := storeDir(t) + dir := storedir.New(t) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() topic := Topic{Tenant: "acme", Table: "t"} diff --git a/internal/mq/mq.go b/internal/mq/mq.go index 089d2b446..f88bcfd54 100644 --- a/internal/mq/mq.go +++ b/internal/mq/mq.go @@ -168,6 +168,20 @@ func WithHeader(key, value string) PublishOpt { } } +// idempotencyHeader carries WithIdempotencyKey's key: JetStream's own +// message-id header, which the stream deduplicates on. +const idempotencyHeader = "Nats-Msg-Id" + +// WithIdempotencyKey marks a publish with key: a second publish carrying the +// same key within the queue's duplicate window is dropped by the broker and +// reported as success, so republishing an event whose first publish had an +// unknown outcome stores it once. +func WithIdempotencyKey(key string) PublishOpt { + return func(h Headers) { + h.Set(idempotencyHeader, key) + } +} + // ErrQueueFull is returned by Publisher.Publish when the queue that holds the // topic's tenant refuses new events because it is at a byte limit — the // backpressure signal the API turns into a 503 with Retry-After. Which limits @@ -178,8 +192,11 @@ var ErrQueueFull = errors.New("ingest queue is full") // ErrUnavailable is returned when the broker cannot be reached or does not // answer in time — a transient failure, not a refusal, that the API turns -// into a 503 with a short Retry-After. Only a backend whose broker is out of -// process returns it; the embedded one's publish failures are plain errors. +// into a 503. Retry-After is the dedupe lease, rounded up to whole seconds, +// when the record held a claim (so an obedient client waits out the window +// instead of retrying straight into it), else a flat few seconds. Only a +// backend whose broker is out of process returns it; the embedded one's +// publish failures are plain errors. var ErrUnavailable = errors.New("message queue unavailable") // Publisher appends events to the ingest queue. diff --git a/internal/mq/mqtest/cases.go b/internal/mq/mqtest/cases.go index 699b51945..513501de8 100644 --- a/internal/mq/mqtest/cases.go +++ b/internal/mq/mqtest/cases.go @@ -157,6 +157,23 @@ func roundTrip(t *testing.T, h Harness) { } } +// A publish repeated with the same idempotency key inside the duplicate +// window is a no-op reported as success: the ingest handler relies on this to +// make a retry of an uncertain publish (the outcome unknown after a failure +// other than a full queue) safe rather than a second copy. A different key +// is its own event. +func idempotencyKeyDropsARepeat(t *testing.T, h Harness) { + b := h.New(t) + topic := mq.Topic{Tenant: Acme, Table: "idem"} + + require.NoError(t, b.Publish(ctx(t), topic, []byte("first"), mq.WithIdempotencyKey("k1"))) + require.NoError(t, b.Publish(ctx(t), topic, []byte("repeat"), mq.WithIdempotencyKey("k1")), + "a repeat under the same key is reported as success, not stored again") + require.NoError(t, b.Publish(ctx(t), topic, []byte("second"), mq.WithIdempotencyKey("k2"))) + + replayEventually(t, b, topic, time.Time{}, []string{"first", "second"}) +} + // Nothing lands on a tenant by omission (#583), and an invalid tenant is not // backpressure a retry could clear. func refusesATopicWithoutATenant(t *testing.T, h Harness) { diff --git a/internal/mq/mqtest/embedded_test.go b/internal/mq/mqtest/embedded_test.go index 98b619d90..f3be090a4 100644 --- a/internal/mq/mqtest/embedded_test.go +++ b/internal/mq/mqtest/embedded_test.go @@ -3,21 +3,19 @@ package mqtest_test import ( - "os" - "path/filepath" "testing" - "time" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/mq/mqtest" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" "github.com/stretchr/testify/require" ) func TestEmbeddedNATS_Conformance(t *testing.T) { mqtest.Run(t, mqtest.Harness{ New: func(t *testing.T) mq.Broker { - e, err := mq.NewEmbedded(storeDir(t)) + e, err := mq.NewEmbedded(storedir.New(t)) require.NoError(t, err) t.Cleanup(func() { _ = e.Close() }) for _, id := range []tenant.ID{mqtest.Acme, mqtest.Globex} { @@ -55,22 +53,3 @@ func TestEmbeddedNATS_Conformance(t *testing.T) { }, }) } - -// storeDir is a temporary store directory whose removal retries briefly: under -// parallel load a consumer's state file can land after Close has returned, -// which fails t.TempDir's one-shot RemoveAll. The retrying cleanup runs first -// (cleanups are LIFO), leaving t.TempDir an empty directory to remove. -func storeDir(t *testing.T) string { - dir := filepath.Join(t.TempDir(), "store") - var err error - t.Cleanup(func() { - for range 50 { - if err = os.RemoveAll(dir); err == nil { - return - } - time.Sleep(20 * time.Millisecond) - } - t.Errorf("remove %s: %v", dir, err) - }) - return dir -} diff --git a/internal/mq/mqtest/mqtest.go b/internal/mq/mqtest/mqtest.go index 5978a5b6a..1d9f2edd9 100644 --- a/internal/mq/mqtest/mqtest.go +++ b/internal/mq/mqtest/mqtest.go @@ -83,6 +83,7 @@ func Run(t *testing.T, h Harness) { cases := []testCase{ {"RoundTrip", true, roundTrip}, + {"IdempotencyKeyDropsARepeat", true, idempotencyKeyDropsARepeat}, {"RefusesATopicWithoutATenant", true, refusesATopicWithoutATenant}, {"SubscribeCarriesTheTraceContext", true, subscribeCarriesTheTraceContext}, {"SubscribeSeesEveryTenant", true, subscribeSeesEveryTenant}, diff --git a/internal/settings/registry_test.go b/internal/settings/registry_test.go index 06e2a3b9a..a5abce849 100644 --- a/internal/settings/registry_test.go +++ b/internal/settings/registry_test.go @@ -79,7 +79,7 @@ func TestRegistry_ReloadWithWarningsAdopts(t *testing.T) { // TestOpen_RejectsInvalid pins the boot contract: an invalid directory yields // no Registry at all — there is no "store without a document" state and no -// compiled defaults to fall back on. +// compiled default for a required key to fall back on. func TestOpen_RejectsInvalid(t *testing.T) { t.Parallel() files := validFiles() @@ -113,9 +113,7 @@ func TestRegistry_SurvivesVanishedDirectory(t *testing.T) { assert.False(t, adopted) assert.True(t, HasErrors(findings)) assert.Equal(t, 42, s.DefaultMaxRows()) - _, id, req := s.DedupeFor("clicks") - assert.Equal(t, "event_id", id) - assert.False(t, req) + assert.Equal(t, "event_id", s.DedupeFor("clicks").IDField) } // TestRegistry_AfterAdoptRunsOnlyOnAdoption pins the lifecycle hook contract diff --git a/internal/settings/seed.go b/internal/settings/seed.go index d9e979ae5..5315c1730 100644 --- a/internal/settings/seed.go +++ b/internal/settings/seed.go @@ -10,7 +10,8 @@ import ( // seedFS holds the starter settings directory: every file present, every // key set to its default. The checked-in seed/ directory is the ONE place -// defaults live — the binary has no compiled fallbacks. Its one consumer is +// defaults live — the binary's one compiled fallback is dedupe.retention +// (missing means "0", forever). Its one consumer is // this embed, so `wavehouse bootstrap` can write the directory anywhere // without a source tree; the container images ship no settings (the operator // mounts or seeds /app/settings), same as they ship no policy file. diff --git a/internal/settings/seed/config.json b/internal/settings/seed/config.json index a8ab41a24..61aa8fee3 100644 --- a/internal/settings/seed/config.json +++ b/internal/settings/seed/config.json @@ -26,6 +26,7 @@ "enabled": false, "id_field": "event_id", "require_id": false, + "retention": "0", "tables": {} }, "dlq": { diff --git a/internal/settings/settings.go b/internal/settings/settings.go index 9ad25b2e7..596318bc4 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -13,6 +13,8 @@ package settings import ( + "time" + "github.com/Wave-RF/WaveHouse/internal/pipes" "github.com/Wave-RF/WaveHouse/internal/policy" ) @@ -65,8 +67,9 @@ type PipesFile struct { // `clickhouse.max_total_conns`), listeners, the observability exporters — // and the secrets (`clickhouse.password`, `auth.jwt_secret`, // `auth.operator_key`), which never belong in a tracked JSON file. Every -// block and every top-level key inside it is REQUIRED: the binary carries no -// compiled defaults, so the adopted snapshot is exactly what the files say. +// block and every top-level key inside it is REQUIRED, but dedupe.retention +// (missing means "0", forever): the binary carries no other compiled default, +// so the adopted snapshot is exactly what the files say. // Defaults live in the seed directory (see Seed) that `wavehouse // bootstrap` writes. The fields are pointers only so Validate can tell // "absent" from the zero value and report it by path. @@ -144,7 +147,8 @@ type AuthConfig struct { // (dedupe.Managed, one per tenant, each a share of the one embedded Pebble // instance), so the whole block is tenant-owned. // -// id_field and require_id are required here and optional per table: a table +// id_field and require_id are required here; retention is optional, and +// missing means "0" (forever). Every field is optional per table: a table // override inherits whichever field it doesn't name. An empty, // whitespace-only, or whitespace-padded id_field is rejected at every level, // so the effective id_field can never be empty or silently unmatchable. @@ -152,6 +156,10 @@ type DedupeConfig struct { Enabled *bool `json:"enabled"` IDField *string `json:"id_field"` RequireID *bool `json:"require_id"` + // Retention is how long a committed id stays a duplicate, as a Go + // duration ("720h"); "0", or leaving it out, keeps it forever. A change + // applies to ids committed after it. + Retention *string `json:"retention,omitempty"` // Tables holds per-table overrides keyed by ClickHouse table name (#222). // Names are format-checked only — existence is schema discovery's runtime // concern, same as policies.json table keys. @@ -163,8 +171,16 @@ type DedupeConfig struct { type TableDedupe struct { IDField *string `json:"id_field,omitempty"` RequireID *bool `json:"require_id,omitempty"` + Retention *string `json:"retention,omitempty"` } +// MinDedupeRetention is the shortest finite dedupe retention: the embedded +// queue's duplicate window (mq.EmbeddedDuplicateWindow). A record is +// published under an idempotency key derived from its id, so an id re-sent +// after a shorter retention but inside the window is claimed again and then +// dropped by the queue as a copy, while the client is told it was accepted. +const MinDedupeRetention = 2 * time.Minute + // DLQConfig gates the Dead Letter Queue: whether a row ClickHouse still // rejects after the row-by-row isolation retry is parked on the tenant's dead-letter // queue (and its original acked) or left unacked to be redelivered diff --git a/internal/settings/store.go b/internal/settings/store.go index f68a2fbf3..a03546411 100644 --- a/internal/settings/store.go +++ b/internal/settings/store.go @@ -16,9 +16,10 @@ import ( // accessors below each resolve from a single snapshot load, so a reload lands // between lookups, never inside one. // -// There are no compiled defaults here on purpose: every key is required by -// Validate, so the snapshot is exactly what the files said when they were -// adopted. Defaults live in the seed directory (Seed / WriteSeed). +// There are no compiled defaults here on purpose, but one: every key but +// dedupe.retention (missing means "0", forever) is required by Validate, so +// the snapshot is exactly what the files said when they were adopted. +// Defaults live in the seed directory (Seed / WriteSeed). type Store struct { // tenant is the id the Registry created the store for; the zero value // only for a Store built outside a Registry (tests). @@ -76,23 +77,41 @@ func (s *Store) DedupeEnabled() bool { return *s.doc().Config.Dedupe.Enabled } +// Dedupe is a table's effective dedupe settings. +type Dedupe struct { + Enabled bool + IDField string + RequireID bool + // Retention is how long a committed id stays a duplicate; 0 is forever. + Retention time.Duration +} + // DedupeFor resolves the effective dedupe settings for a table: the switch, // then the table override for each field it names, the global value -// otherwise. All three resolve from one snapshot load, so a reload can never -// hand a record the id_field of one document and the require_id (or enabled) -// of another. -func (s *Store) DedupeFor(table string) (enabled bool, idField string, requireID bool) { +// otherwise. Every field resolves from one snapshot load, so a reload can +// never hand a record the id_field of one document and the require_id, +// retention or switch of another. +func (s *Store) DedupeFor(table string) Dedupe { d := s.doc().Config.Dedupe - enabled, idField, requireID = *d.Enabled, *d.IDField, *d.RequireID + out := Dedupe{Enabled: *d.Enabled, IDField: *d.IDField, RequireID: *d.RequireID} + retention := "0" + if d.Retention != nil { + retention = *d.Retention + } if td, ok := d.Tables[table]; ok { if td.IDField != nil { - idField = *td.IDField + out.IDField = *td.IDField } if td.RequireID != nil { - requireID = *td.RequireID + out.RequireID = *td.RequireID + } + if td.Retention != nil { + retention = *td.Retention } } - return enabled, idField, requireID + // Validate has parsed it already. + out.Retention, _ = time.ParseDuration(retention) + return out } // ClickHouse is the adopted connection wiring, resolved as one value from diff --git a/internal/settings/store_test.go b/internal/settings/store_test.go index abcc6c902..3487ba505 100644 --- a/internal/settings/store_test.go +++ b/internal/settings/store_test.go @@ -1,6 +1,7 @@ package settings import ( + "encoding/json" "os" "path/filepath" "testing" @@ -46,27 +47,42 @@ func TestStore_Tenant(t *testing.T) { func TestStore_DedupeFor_Cascade(t *testing.T) { t.Parallel() s := newLoadedStore(t, map[string]string{ - FileConfig: configJSON(`{"dedupe": {"require_id": true, "tables": {"clicks": {"id_field": "click_id"}, "views": {"require_id": false}}}}`), + FileConfig: configJSON(`{"dedupe": {"require_id": true, "retention": "720h", "tables": {"clicks": {"id_field": "click_id"}, "views": {"require_id": false, "retention": "24h"}, "audit": {"retention": "0"}}}}`), }) tests := []struct { - name, table, wantID string - wantRequire bool + name, table string + want Dedupe }{ - {name: "table overrides id_field, inherits require_id", table: "clicks", wantID: "click_id", wantRequire: true}, - {name: "table overrides require_id, inherits id_field", table: "views", wantID: "event_id", wantRequire: false}, - {name: "unlisted table gets globals", table: "other", wantID: "event_id", wantRequire: true}, + {name: "table overrides id_field, inherits the rest", table: "clicks", want: Dedupe{IDField: "click_id", RequireID: true, Retention: 720 * time.Hour}}, + {name: "table overrides require_id and retention, inherits id_field", table: "views", want: Dedupe{IDField: "event_id", Retention: 24 * time.Hour}}, + {name: "table keeps ids forever under a finite tenant retention", table: "audit", want: Dedupe{IDField: "event_id", RequireID: true}}, + {name: "unlisted table gets globals", table: "other", want: Dedupe{IDField: "event_id", RequireID: true, Retention: 720 * time.Hour}}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { t.Parallel() - _, id, req := s.DedupeFor(tt.table) - assert.Equal(t, tt.wantID, id) - assert.Equal(t, tt.wantRequire, req) + assert.Equal(t, tt.want, s.DedupeFor(tt.table)) }) } } +// A config.json without dedupe.retention keeps ids forever, and its table +// overrides inherit that or set their own. +func TestStore_DedupeFor_RetentionMissing(t *testing.T) { + t.Parallel() + var doc map[string]map[string]any + require.NoError(t, json.Unmarshal([]byte(configJSON(`{"dedupe": {"tables": {"clicks": {"id_field": "click_id"}, "views": {"retention": "24h"}}}}`)), &doc)) + delete(doc["dedupe"], "retention") + body, err := json.Marshal(doc) + require.NoError(t, err) + s := newLoadedStore(t, map[string]string{FileConfig: string(body)}) + + assert.Equal(t, Dedupe{IDField: "event_id"}, s.DedupeFor("other"), "forever") + assert.Equal(t, Dedupe{IDField: "click_id"}, s.DedupeFor("clicks"), "inherits forever") + assert.Equal(t, Dedupe{IDField: "event_id", Retention: 24 * time.Hour}, s.DedupeFor("views")) +} + // TestStore_SeedIsValid pins that the shipped starter directory passes its // own gate: `wavehouse bootstrap` must never write something // `wavehouse validate` rejects, and the defaults are readable back. @@ -83,9 +99,7 @@ func TestStore_SeedIsValid(t *testing.T) { // decision (deployments/compose/settings ships the opt-in trial one). assert.Len(t, findings, 1, "findings: %s", findingStrings(findings)) assert.Contains(t, findingStrings(findings), "no policy") - _, id, req := s.DedupeFor("anything") - assert.Equal(t, "event_id", id) - assert.False(t, req) + assert.Equal(t, Dedupe{IDField: "event_id"}, s.DedupeFor("anything"), "retention 0: ids kept forever, as before retention existed") assert.Equal(t, ClickHouse{Addr: "localhost:9000", HTTPPort: 8123, HTTPScheme: "http", Database: "default", Username: "default", QueryTimeout: 30 * time.Second, Headers: map[string]string{}, MaxOpenConns: 10, MaxIdleConns: 5}, s.ClickHouse()) assert.Equal(t, Auth{JWKSURL: "", RoleClaim: "role"}, s.Auth()) assert.True(t, s.DLQFor("anything")) diff --git a/internal/settings/validate.go b/internal/settings/validate.go index 598c4aa7d..b7cf8a52c 100644 --- a/internal/settings/validate.go +++ b/internal/settings/validate.go @@ -12,6 +12,7 @@ import ( "path/filepath" "slices" "strings" + "time" "github.com/Wave-RF/WaveHouse/internal/pipes" "github.com/Wave-RF/WaveHouse/internal/policy" @@ -450,6 +451,26 @@ func (v *validator) checkIDField(path string, val *string) { } } +// checkRetention rejects a dedupe retention that is not a duration, is +// negative, or is finite but shorter than MinDedupeRetention. The short one +// is refused rather than raised to the minimum, so the file never means +// something other than what it says. nil is valid: forever at the tenant +// level, inherited at the table level. +func (v *validator) checkRetention(path string, val *string) { + if val == nil { + return + } + d, err := time.ParseDuration(*val) + switch { + case err != nil: + v.errorf(FileConfig, path, "must be a duration such as \"720h\", or \"0\" to keep ids forever, got %q", *val) + case d < 0: + v.errorf(FileConfig, path, "must not be negative, got %q", *val) + case d > 0 && d < MinDedupeRetention: + v.errorf(FileConfig, path, "%q is shorter than the ingest queue's %s duplicate window: an id re-sent after it expires but inside the window would be dropped by the queue while the client is told it was accepted — use at least %q, or \"0\" to keep ids forever", *val, MinDedupeRetention, MinDedupeRetention.String()) + } +} + // checkTableName rejects a per-table override key that could never match a // table: empty, or carrying surrounding whitespace. Shared by the dedupe and // dlq override maps. @@ -670,14 +691,16 @@ func (v *validator) parseConfig(data []byte) TenantConfig { v.required("dedupe.require_id") } v.checkIDField("dedupe.id_field", d.IDField) + v.checkRetention("dedupe.retention", d.Retention) // Sorted iteration keeps finding order deterministic across runs. for _, table := range slices.Sorted(maps.Keys(d.Tables)) { td := d.Tables[table] path := "dedupe.tables." + table v.checkTableName("dedupe.tables", table) v.checkIDField(path+".id_field", td.IDField) - if td.IDField == nil && td.RequireID == nil { - v.warnf(FileConfig, path, "override sets nothing — remove it, or set id_field or require_id") + v.checkRetention(path+".retention", td.Retention) + if td.IDField == nil && td.RequireID == nil && td.Retention == nil { + v.warnf(FileConfig, path, "override sets nothing — remove it, or set id_field, require_id or retention") } } } diff --git a/internal/settings/validate_test.go b/internal/settings/validate_test.go index 56be6b2a0..c7a80470e 100644 --- a/internal/settings/validate_test.go +++ b/internal/settings/validate_test.go @@ -33,8 +33,9 @@ func validFiles() map[string]string { // configJSON returns the seed config.json with patch merged over it, one // level deep (a patched block's keys replace the seed's, the rest of the -// block is kept). Every key is required, so tests that care about one key -// build a complete document from the seed rather than repeating all of them. +// block is kept). Every key but dedupe.retention is required, so tests that +// care about one key build a complete document from the seed rather than +// repeating all of them. func configJSON(patch string) string { seed, err := Seed() if err != nil { @@ -268,13 +269,13 @@ func TestValidate_ContentRules(t *testing.T) { {"negative max rows", FileConfig, `{"query": {"default_max_rows": -1}}`, "must be >= 1"}, {"zero max rows", FileConfig, `{"query": {"default_max_rows": 0}}`, "must be >= 1"}, {"missing dedupe block", FileConfig, `{"dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "dedupe: required"}, - {"missing dlq block", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "dlq: required"}, + {"missing dlq block", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "dlq: required"}, {"missing dlq.enabled", FileConfig, `{"dlq": {"tables": {}}}`, "dlq.enabled: required"}, {"empty dlq override table name", FileConfig, `{"dlq": {"tables": {"": {"enabled": false}}}}`, "table name must not be empty"}, {"dlq override table whitespace", FileConfig, `{"dlq": {"tables": {"clicks ": {"enabled": false}}}}`, "surrounding whitespace"}, {"missing query.timestamp_bucket_seconds", FileConfig, `{"query": {"default_max_rows": 1}}`, "query.timestamp_bucket_seconds: required"}, {"negative timestamp bucket", FileConfig, `{"query": {"timestamp_bucket_seconds": -1}}`, "must be >= 0"}, - {"missing stream block", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "cors": {"allowed_origins": []}}`, "stream: required"}, + {"missing stream block", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "cors": {"allowed_origins": []}}`, "stream: required"}, {"missing stream.keepalive_interval", FileConfig, `{"stream": {"keepalive_buckets": 3, "gap_window_minutes": 15}}`, "stream.keepalive_interval: required"}, {"missing stream.keepalive_buckets", FileConfig, `{"stream": {"keepalive_interval": 30, "gap_window_minutes": 15}}`, "stream.keepalive_buckets: required"}, {"missing stream.gap_window_minutes", FileConfig, `{"stream": {"keepalive_interval": 30, "keepalive_buckets": 3}}`, "stream.gap_window_minutes: required"}, @@ -283,14 +284,20 @@ func TestValidate_ContentRules(t *testing.T) { {"negative gap window", FileConfig, `{"stream": {"gap_window_minutes": -1}}`, "stream.gap_window_minutes: must be >= 0"}, {"keepalive as a duration string", FileConfig, `{"stream": {"keepalive_interval": "30s"}}`, "keepalive_interval"}, {"missing dedupe.require_id", FileConfig, `{"dedupe": {"id_field": "event_id"}}`, "dedupe.require_id: required"}, - {"missing dedupe.enabled", FileConfig, `{"dedupe": {"id_field": "event_id", "require_id": false}}`, "dedupe.enabled: required"}, + {"missing dedupe.enabled", FileConfig, `{"dedupe": {"id_field": "event_id", "require_id": false, "retention": "0"}}`, "dedupe.enabled: required"}, + {"dedupe.retention not a duration", FileConfig, configJSON(`{"dedupe": {"retention": "30d"}}`), `dedupe.retention: must be a duration such as "720h"`}, + {"dedupe.retention a number", FileConfig, configJSON(`{"dedupe": {"retention": 3600}}`), "retention"}, + {"dedupe.retention negative", FileConfig, configJSON(`{"dedupe": {"retention": "-1h"}}`), "dedupe.retention: must not be negative"}, + {"dedupe.retention under the duplicate window", FileConfig, configJSON(`{"dedupe": {"retention": "1m59s"}}`), `dedupe.retention: "1m59s" is shorter than the ingest queue's 2m0s duplicate window`}, + {"override retention under the duplicate window", FileConfig, configJSON(`{"dedupe": {"tables": {"clicks": {"retention": "30s"}}}}`), "dedupe.tables.clicks.retention: \"30s\" is shorter"}, + {"override retention not a duration", FileConfig, configJSON(`{"dedupe": {"tables": {"clicks": {"retention": "forever"}}}}`), "dedupe.tables.clicks.retention: must be a duration"}, {"missing query.default_max_rows", FileConfig, `{"query": {}}`, "query.default_max_rows: required"}, {"missing schema.refresh_interval", FileConfig, `{"schema": {}}`, "schema.refresh_interval: required"}, {"missing cors.allowed_origins", FileConfig, `{"cors": {}}`, "cors.allowed_origins: required"}, {"empty config document", FileConfig, `{}`, "cors: required"}, {"otel is boot config", FileConfig, `{"otel": {"enabled": true}}`, "unknown field"}, - {"missing clickhouse block", FileConfig, `{"auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "clickhouse: required"}, - {"missing auth block", FileConfig, `{"clickhouse": {"addr": "h:9000", "http_port": 8123, "http_scheme": "http", "database": "d", "username": "u", "query_timeout": 1}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "auth: required"}, + {"missing clickhouse block", FileConfig, `{"auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "clickhouse: required"}, + {"missing auth block", FileConfig, `{"clickhouse": {"addr": "h:9000", "http_port": 8123, "http_scheme": "http", "database": "d", "username": "u", "query_timeout": 1}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "auth: required"}, {"missing clickhouse.addr", FileConfig, `{"clickhouse": {"http_port": 8123, "http_scheme": "http", "database": "d", "username": "u", "query_timeout": 1}}`, "clickhouse.addr: required"}, {"clickhouse.addr without port", FileConfig, `{"clickhouse": {"addr": "localhost"}}`, "must be host:port"}, {"clickhouse.http_port out of range", FileConfig, `{"clickhouse": {"http_port": 70000}}`, "clickhouse.http_port: must be in 1-65535"}, @@ -343,6 +350,67 @@ func TestValidate_ContentRules(t *testing.T) { } } +// A finite retention at or above the duplicate window is accepted, "0" (or +// any zero duration) is forever, and a table may keep ids longer or shorter +// than the tenant, or forever under a finite tenant retention. +func TestValidate_DedupeRetentionAccepted(t *testing.T) { + t.Parallel() + for _, patch := range []string{ + `{"dedupe": {"retention": "0"}}`, + `{"dedupe": {"retention": "0s"}}`, + `{"dedupe": {"retention": "2m"}}`, + `{"dedupe": {"retention": "720h", "tables": {"clicks": {"retention": "24h"}, "views": {"retention": "0"}}}}`, + } { + t.Run(patch, func(t *testing.T) { + t.Parallel() + files := validFiles() + files[FileConfig] = configJSON(patch) + doc, findings := ValidateDir(writeDir(t, files)) + require.NotNil(t, doc, "findings: %s", findingStrings(findings)) + assert.False(t, HasErrors(findings)) + }) + } +} + +// configJSONWithout is the seed config.json less dedupe.retention. +func configJSONWithout(t *testing.T) string { + t.Helper() + var doc map[string]map[string]json.RawMessage + require.NoError(t, json.Unmarshal([]byte(configJSON(`{}`)), &doc)) + delete(doc["dedupe"], "retention") + out, err := json.Marshal(doc) + require.NoError(t, err) + return string(out) +} + +// dedupe.retention may be left out: the tenant keeps ids forever, and a +// table override may still set one. +func TestValidate_DedupeRetentionOptional(t *testing.T) { + t.Parallel() + files := validFiles() + files[FileConfig] = configJSONWithout(t) + require.NotContains(t, files[FileConfig], "retention") + doc, findings := ValidateDir(writeDir(t, files)) + require.NotNil(t, doc, "findings: %s", findingStrings(findings)) + assert.False(t, HasErrors(findings), "findings: %s", findingStrings(findings)) + assert.Nil(t, doc.Config.Dedupe.Retention) +} + +// A table name with odd bytes — NUL included — is any other table name to +// the override maps: dedupe keys escape the table (internal/keyenc), so any +// bytes are just another table name. +func TestValidate_OverrideTableNamesAnyBytes(t *testing.T) { + t.Parallel() + files := validFiles() + files[FileConfig] = configJSON(`{"dedupe": {"tables": {"cli\u0000cks": {"require_id": true}, "a\u0001b\tc": {"id_field": "x"}}}, "dlq": {"tables": {"cli\u0000cks": {"enabled": false}}}}`) + doc, findings := ValidateDir(writeDir(t, files)) + require.Empty(t, findings, "findings: %s", findingStrings(findings)) + require.NotNil(t, doc) + assert.Contains(t, doc.Config.Dedupe.Tables, "cli\x00cks") + assert.Contains(t, doc.Config.Dedupe.Tables, "a\x01b\tc") + assert.Contains(t, doc.Config.DLQ.Tables, "cli\x00cks") +} + // TestValidate_ClickHouseTLSPathsAreNotOpened pins that the tls block is // checked for shape only: Validate is pure and also runs on the control // plane, so paths that exist nowhere still validate, and the values reach diff --git a/internal/testutil/mocks.go b/internal/testutil/mocks.go index 31ce6883f..67c2beb04 100644 --- a/internal/testutil/mocks.go +++ b/internal/testutil/mocks.go @@ -3,6 +3,7 @@ package testutil import ( "bytes" "context" + "fmt" "io" "net/http" "sync" @@ -10,6 +11,7 @@ import ( "time" "github.com/Wave-RF/WaveHouse/internal/cache" + "github.com/Wave-RF/WaveHouse/internal/dedupe" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/tenant" ) @@ -33,6 +35,10 @@ type MockPublisher struct { mu sync.Mutex Messages []PublishedMessage Err error // if set, Publish and DeadLetter return this error + // ErrAfter lets that many calls succeed before Err applies, to fail a + // batch part-way through. + ErrAfter int + calls int } // PublishedMessage records a single Publish or DeadLetter call, with the @@ -53,15 +59,16 @@ func (m *MockPublisher) DeadLetter(_ context.Context, msg *mq.Message, opts ...m } func (m *MockPublisher) record(pm PublishedMessage, opts []mq.PublishOpt) error { - if m.Err != nil { + m.mu.Lock() + defer m.mu.Unlock() + m.calls++ + if m.Err != nil && m.calls > m.ErrAfter { return m.Err } headers := mq.Headers{} for _, opt := range opts { opt(headers) } - m.mu.Lock() - defer m.mu.Unlock() pm.Headers = headers m.Messages = append(m.Messages, pm) return nil @@ -104,28 +111,118 @@ func (m *MockSubscriber) Close() error { return nil } // ── Mock Deduplicator ──────────────────────────────────────────── -// MockDeduplicator implements dedupe.Deduplicator for testing. +// MockDeduplicator implements dedupe.Deduplicator in memory, with per-phase +// error injection and a record of what was committed and released. type MockDeduplicator struct { - mu sync.Mutex - seen map[string]bool - Err error // if set, CheckAndMark returns this error + mu sync.Mutex + committed map[dedupe.Key]bool + retention map[dedupe.Key]time.Duration // each commit's retention + pending map[dedupe.Key]string + tokens int + // Err, if set, fails Reserve — after ErrAfter calls have succeeded; + // CommitErr and ReleaseErr fail their phase. + Err error + ErrAfter int + CommitErr error + ReleaseErr error + Released []dedupe.Claim // every claim Release was given + // Calls to each phase, for tests that count round trips. + Reserves, Commits int } +var _ dedupe.Deduplicator = (*MockDeduplicator)(nil) + func NewMockDeduplicator() *MockDeduplicator { - return &MockDeduplicator{seen: make(map[string]bool)} + return &MockDeduplicator{committed: map[dedupe.Key]bool{}, retention: map[dedupe.Key]time.Duration{}, pending: map[dedupe.Key]string{}} +} + +// Reserve answers Duplicate for a key repeated in one call, as Managed does. +func (m *MockDeduplicator) Reserve(_ context.Context, keys []dedupe.Key, _ time.Duration) ([]dedupe.Claim, error) { + m.mu.Lock() + defer m.mu.Unlock() + m.Reserves++ + if m.Err != nil && m.Reserves > m.ErrAfter { + return nil, m.Err + } + claims := make([]dedupe.Claim, 0, len(keys)) + seen := make(map[dedupe.Key]bool, len(keys)) + for _, k := range keys { + repeat := seen[k] + seen[k] = true + switch { + case repeat, m.committed[k]: + claims = append(claims, dedupe.Claim{Key: k, Status: dedupe.Duplicate}) + case m.pending[k] != "": + claims = append(claims, dedupe.Claim{Key: k, Status: dedupe.InFlight}) + default: + m.tokens++ + tok := fmt.Sprint(m.tokens) + m.pending[k] = tok + claims = append(claims, dedupe.Claim{Key: k, Status: dedupe.Claimed, Token: tok}) + } + } + return claims, nil } -func (m *MockDeduplicator) CheckAndMark(_ context.Context, eventID string) (bool, error) { - if m.Err != nil { - return false, m.Err +func (m *MockDeduplicator) Commit(_ context.Context, claims []dedupe.Claim, retention time.Duration) error { + m.mu.Lock() + defer m.mu.Unlock() + m.Commits++ + if m.CommitErr != nil { + return m.CommitErr + } + for _, c := range claims { + if c.Status == dedupe.Claimed { + m.committed[c.Key] = true + m.retention[c.Key] = retention + delete(m.pending, c.Key) + } } + return nil +} + +func (m *MockDeduplicator) Release(_ context.Context, claims []dedupe.Claim) error { m.mu.Lock() defer m.mu.Unlock() - if m.seen[eventID] { - return true, nil + m.Released = append(m.Released, claims...) + if m.ReleaseErr != nil { + return m.ReleaseErr + } + for _, c := range claims { + if c.Status == dedupe.Claimed && m.pending[c.Key] == c.Token { + delete(m.pending, c.Key) + } } - m.seen[eventID] = true - return false, nil + return nil +} + +// Hold claims k as another in-flight request would, so Reserve answers +// InFlight for it. +func (m *MockDeduplicator) Hold(k dedupe.Key) { + m.mu.Lock() + defer m.mu.Unlock() + m.pending[k] = "held" +} + +// Committed reports whether k was committed. +func (m *MockDeduplicator) Committed(k dedupe.Key) bool { + m.mu.Lock() + defer m.mu.Unlock() + return m.committed[k] +} + +// Retention is the retention k was last committed with. +func (m *MockDeduplicator) Retention(k dedupe.Key) time.Duration { + m.mu.Lock() + defer m.mu.Unlock() + return m.retention[k] +} + +// Pending reports whether k is claimed and neither committed nor released. +func (m *MockDeduplicator) Pending(k dedupe.Key) bool { + m.mu.Lock() + defer m.mu.Unlock() + return m.pending[k] != "" } func (m *MockDeduplicator) Close() error { return nil } diff --git a/internal/testutil/storedir/storedir.go b/internal/testutil/storedir/storedir.go new file mode 100644 index 000000000..aae618636 --- /dev/null +++ b/internal/testutil/storedir/storedir.go @@ -0,0 +1,47 @@ +// Package storedir gives a test a directory for the embedded message broker's +// store. It imports nothing from the repository, so internal/mq's own tests can +// use it. +package storedir + +import ( + "errors" + "os" + "path/filepath" + "syscall" + "testing" +) + +// maxRemovals bounds the store's removal: a consumer-state write still under way +// when the broker closes adds at most two entries after it — its temporary file, +// then the rename into place — so a removal is refilled at most twice per +// durable. Past this many, something is writing that Close did not stop. +const maxRemovals = 32 + +// New returns an empty directory under t.TempDir for a broker's store, removed +// once the broker is closed: New's cleanup runs after the test's own, Close +// among them (cleanups run last-in, first-out). +// +// A closed broker's store is not yet quiescent: the embedded NATS server writes +// each durable consumer's state from a goroutine its Shutdown does not join, so +// a write under way can land after Close has returned — failing t.TempDir's +// one-shot RemoveAll with "directory not empty" (#442). There is nothing to +// wait on, so the removal is tried again whenever a directory was refilled +// between reading and removing it. That ends without a clock: those writes add a +// bounded number of entries, and none once their directory is gone. +func New(t testing.TB) string { + t.Helper() + dir := filepath.Join(t.TempDir(), "store") + if err := os.Mkdir(dir, 0o700); err != nil { + t.Fatalf("store directory: %v", err) + } + t.Cleanup(func() { + err := os.RemoveAll(dir) + for i := 1; i < maxRemovals && errors.Is(err, syscall.ENOTEMPTY); i++ { + err = os.RemoveAll(dir) + } + if err != nil { + t.Errorf("remove the store: %v", err) + } + }) + return dir +} diff --git a/internal/testutil/storedir/storedir_test.go b/internal/testutil/storedir/storedir_test.go new file mode 100644 index 000000000..6720a5349 --- /dev/null +++ b/internal/testutil/storedir/storedir_test.go @@ -0,0 +1,28 @@ +package storedir + +import ( + "os" + "path/filepath" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestNew_RemovedAfterTheTestsOwnCleanups(t *testing.T) { + var dir string + t.Run("store", func(t *testing.T) { + dir = New(t) + entries, err := os.ReadDir(dir) + require.NoError(t, err) + assert.Empty(t, entries, "the store starts empty") + // Registered after New, as a broker's Close is, so it runs first and + // what it writes is removed with the rest. + t.Cleanup(func() { + obs := filepath.Join(dir, "jetstream", "obs") + require.NoError(t, os.MkdirAll(obs, 0o700)) + require.NoError(t, os.WriteFile(filepath.Join(obs, "o.dat"), nil, 0o600)) + }) + }) + assert.NoDirExists(t, dir) +} diff --git a/internal/testutil/testutil.go b/internal/testutil/testutil.go index db1196869..8b3cefb4b 100644 --- a/internal/testutil/testutil.go +++ b/internal/testutil/testutil.go @@ -16,6 +16,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" ) // NewTestSchemaRegistry creates a SchemaRegistry pre-loaded with the given @@ -40,13 +41,13 @@ func NewTestSchemaRegistry(t testing.TB, tables []*discovery.TableSchema) *disco // hardcoding the same literal twice. const TestServerVersion = "24.8.1.1" -// NewEmbeddedMQ starts the embedded broker over a temporary directory, closed -// by the test framework, with a queue open for each of tenants — +// NewEmbeddedMQ starts the embedded broker over a storedir.New directory, +// closed by the test framework, with a queue open for each of tenants — // tenant.Default when none is named — at maxBytes: a tenant has a queue once // its budget is applied, as the wiring does for every tenant it serves. func NewEmbeddedMQ(t testing.TB, maxBytes int64, tenants ...tenant.ID) *mq.EmbeddedNATS { t.Helper() - emb, err := mq.NewEmbedded(t.TempDir()) + emb, err := mq.NewEmbedded(storedir.New(t)) require.NoError(t, err) t.Cleanup(func() { _ = emb.Close() }) if len(tenants) == 0 { diff --git a/tests/e2e/fixtures/settings/config.json b/tests/e2e/fixtures/settings/config.json index a0d15cfce..a8d7c7367 100644 --- a/tests/e2e/fixtures/settings/config.json +++ b/tests/e2e/fixtures/settings/config.json @@ -26,6 +26,7 @@ "enabled": true, "id_field": "event_id", "require_id": false, + "retention": "0", "tables": {} }, "dlq": { diff --git a/tests/integration/dedupe_dynamodb_app_test.go b/tests/integration/dedupe_dynamodb_app_test.go new file mode 100644 index 000000000..50edcf018 --- /dev/null +++ b/tests/integration/dedupe_dynamodb_app_test.go @@ -0,0 +1,170 @@ +//go:build integration + +package tests + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net" + "net/http" + "net/url" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/app" + "github.com/Wave-RF/WaveHouse/internal/config" + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// TestDynamoDBDedupe_TwoInstancesShareSeenIDs boots two apps the way two +// pods run — each its own data_dir, embedded queue and ingest worker — with +// dedupe.backend dynamodb over one table on dynamodb-local, and checks an id +// ingested through either is a duplicate through the other, that ClickHouse +// holds each id once, and that dedupe.lease reaches ingest as the in-flight +// answer's Retry-After. +func TestDynamoDBDedupe_TwoInstancesShareSeenIDs(t *testing.T) { + e := env(t) + ctx := context.Background() + // The SDK's default chain, as in production; never the developer's files. + none := filepath.Join(t.TempDir(), "none") + for k, v := range map[string]string{ + "AWS_ACCESS_KEY_ID": "local", "AWS_SECRET_ACCESS_KEY": "local", "AWS_SESSION_TOKEN": "", + "AWS_PROFILE": "", "AWS_CONFIG_FILE": none, "AWS_SHARED_CREDENTIALS_FILE": none, + "AWS_EC2_METADATA_DISABLED": "true", + } { + t.Setenv(k, v) + } + + chTable := createTable(t, "event_id String, n UInt32", "ORDER BY event_id") + ddbTable := newDynamoTable() + const lease = 7 * time.Second + + boot := func(name string) string { + t.Helper() + files, err := tenantSettings(e.ch, testCHDatabase) + require.NoError(t, err) + var doc map[string]json.RawMessage + require.NoError(t, json.Unmarshal(files[settings.FileConfig], &doc)) + doc["dedupe"] = json.RawMessage(`{"enabled": true, "id_field": "event_id", "require_id": true, "tables": {}}`) + files[settings.FileConfig], err = json.Marshal(doc) + require.NoError(t, err) + dir := filepath.Join(t.TempDir(), name) + require.NoError(t, writeSettingsFiles(dir, files)) + + var lc net.ListenConfig + ln, err := lc.Listen(ctx, "tcp", "127.0.0.1:0") + require.NoError(t, err) + cfg := &config.Config{ + DataDir: t.TempDir(), + Server: config.Server{ShutdownTimeout: 10}, + ClickHouse: config.ClickHouse{Password: testCHPassword}, + MQ: config.MQ{Backend: config.MQEmbedded}, + Cache: config.Cache{Backend: config.CacheLocal, L1MaxCost: 1 << 20}, + Dedupe: config.Dedupe{Backend: config.DedupeDynamoDB, Lease: lease, DynamoDB: config.DedupeDynamoDBConfig{ + Table: ddbTable, Region: "us-east-1", Endpoint: e.dynamoEndpoint, + // dynamodb-local under a parallel suite is slower than the real thing. + Timeout: 5 * time.Second, CreateTable: true, + }}, + Coord: config.Coord{Backend: config.CoordLocal}, + Roles: config.AllRoles(), + Settings: config.Settings{Dir: dir}, + } + a, err := app.New(ctx, app.Options{Config: cfg, Listener: ln}) + require.NoError(t, err) + runCtx, stop := context.WithCancel(ctx) + runDone := make(chan error, 1) + go func() { runDone <- a.Run(runCtx) }() + t.Cleanup(func() { + stop() + assert.NoError(t, <-runDone) + closeCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + assert.NoError(t, a.Close(closeCtx)) + }) + baseURL := "http://" + ln.Addr().String() + require.NoError(t, waitForLive(ctx, baseURL, 30*time.Second)) + return baseURL + } + // Both create the table: the second finds it and leaves it as it is. + podA, podB := boot("a"), boot("b") + + ingest := func(baseURL, id string, n int) (int, string, http.Header) { + t.Helper() + body := fmt.Sprintf(`{"event_id": %q, "n": %d}`, id, n) + req, err := http.NewRequestWithContext(ctx, http.MethodPost, baseURL+"/v1/ingest?table="+url.QueryEscape(chTable), strings.NewReader(body)) + require.NoError(t, err) + req.Header.Set("Content-Type", "application/json") + resp, err := http.DefaultClient.Do(req) + require.NoError(t, err) + defer func() { _ = resp.Body.Close() }() + b, err := io.ReadAll(resp.Body) + require.NoError(t, err) + return resp.StatusCode, strings.TrimSpace(string(b)), resp.Header + } + accepted := func(baseURL, id string, n int) { + t.Helper() + status, body, _ := ingest(baseURL, id, n) + require.Equal(t, http.StatusOK, status, body) + require.JSONEq(t, `{"ok": true}`, body) + } + duplicate := func(baseURL, id string, n int) { + t.Helper() + status, body, _ := ingest(baseURL, id, n) + require.Equal(t, http.StatusOK, status, body) + require.JSONEq(t, `{"duplicate": true}`, body) + } + + accepted(podA, "e1", 1) + duplicate(podB, "e1", 2) + accepted(podB, "e2", 3) + duplicate(podA, "e2", 4) + duplicate(podA, "e1", 5) + + // A claim another process holds is in flight on both pods, for as long + // as the configured lease says. + peer := dynamoClient(t, ddbTable, dedupe.DynamoConfig{}).Tenant(tenant.Default) + require.NoError(t, peer.Apply(true)) + claims, err := peer.Reserve(ctx, []dedupe.Key{{Table: chTable, ID: "e3"}}, time.Minute) + require.NoError(t, err) + require.Equal(t, dedupe.Claimed, claims[0].Status) + for _, pod := range []string{podA, podB} { + status, body, header := ingest(pod, "e3", 6) + require.Equal(t, http.StatusServiceUnavailable, status, body) + assert.Equal(t, "7", header.Get("Retry-After"), "dedupe.lease, in seconds") + } + require.NoError(t, peer.Release(ctx, claims)) + accepted(podB, "e3", 7) + duplicate(podA, "e3", 8) + + // Each pod's worker wrote only what its pod accepted: each id once. + type row struct { + ID string + N uint32 + } + want := []row{{"e1", 1}, {"e2", 3}, {"e3", 7}} + require.Eventually(t, func() bool { + rows, err := e.chConn.Query(ctx, fmt.Sprintf("SELECT event_id, n FROM %s ORDER BY event_id", chTable)) + if err != nil { + return false + } + defer func() { _ = rows.Close() }() + var got []row + for rows.Next() { + var r row + if rows.Scan(&r.ID, &r.N) != nil { + return false + } + got = append(got, r) + } + return assert.ObjectsAreEqual(want, got) + }, 30*time.Second, 500*time.Millisecond, "ClickHouse holds each id once, from the pod that accepted it") +} diff --git a/tests/integration/dedupe_dynamodb_test.go b/tests/integration/dedupe_dynamodb_test.go new file mode 100644 index 000000000..a3cf576b6 --- /dev/null +++ b/tests/integration/dedupe_dynamodb_test.go @@ -0,0 +1,348 @@ +//go:build integration + +package tests + +import ( + "context" + "errors" + "fmt" + "io" + "net/http" + "strconv" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/aws/aws-sdk-go-v2/aws" + "github.com/aws/aws-sdk-go-v2/config" + "github.com/aws/aws-sdk-go-v2/credentials" + "github.com/aws/aws-sdk-go-v2/service/dynamodb" + "github.com/aws/aws-sdk-go-v2/service/dynamodb/types" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/dedupe/dedupetest" +) + +var dynamoTables atomic.Uint64 + +// newDynamoTable names a fresh table on dynamodb-local for one test. +func newDynamoTable() string { + return fmt.Sprintf("dedupe_%d", dynamoTables.Add(1)) +} + +// dynamoClient is one client — one pod's view — over table on +// dynamodb-local, through the production constructor. +func dynamoClient(t *testing.T, table string, cfg dedupe.DynamoConfig, extra ...func(*config.LoadOptions) error) *dedupe.Dynamo { + t.Helper() + cfg.Table, cfg.Endpoint, cfg.Region = table, env(t).dynamoEndpoint, "us-east-1" + if cfg.Timeout == 0 { + // dynamodb-local under a parallel suite is slower than the real thing. + cfg.Timeout = 5 * time.Second + } + opts := append([]func(*config.LoadOptions) error{ + config.WithCredentialsProvider(credentials.NewStaticCredentialsProvider("local", "local", "")), + }, extra...) + d, err := dedupe.NewDynamo(t.Context(), cfg, opts...) + require.NoError(t, err) + return d +} + +// rawDynamo is a plain client, for reading and planting items directly. +func rawDynamo(t *testing.T) *dynamodb.Client { + t.Helper() + return dynamodb.New(dynamodb.Options{ + Region: "us-east-1", + BaseEndpoint: aws.String(env(t).dynamoEndpoint), + Credentials: credentials.NewStaticCredentialsProvider("local", "local", ""), + }) +} + +// faultyHTTP answers matching requests itself instead of sending them. +type faultyHTTP struct { + next *http.Client + fault func(target string) (*http.Response, error, bool) +} + +func (f *faultyHTTP) Do(r *http.Request) (*http.Response, error) { + if resp, err, ok := f.fault(r.Header.Get("X-Amz-Target")); ok { + return resp, err + } + return f.next.Do(r) +} + +// awsError is a DynamoDB JSON error response. +func awsError(status int, code string) *http.Response { + body := fmt.Sprintf(`{"__type":"com.amazonaws.dynamodb.v20120810#%s","message":"injected"}`, code) + return &http.Response{ + StatusCode: status, + Header: http.Header{"Content-Type": {"application/x-amz-json-1.0"}}, + Body: io.NopCloser(strings.NewReader(body)), + } +} + +const putItem = "DynamoDB_20120810.PutItem" + +// landThenFail sends the first PutItem, then answers it 500 as if the +// response were lost: the write is applied and the SDK retries it. +type landThenFail struct { + next *http.Client + failed atomic.Bool +} + +func (l *landThenFail) Do(r *http.Request) (*http.Response, error) { + resp, err := l.next.Do(r) + if err != nil || r.Header.Get("X-Amz-Target") != putItem || !l.failed.CompareAndSwap(false, true) { + return resp, err + } + _, _ = io.Copy(io.Discard, resp.Body) + _ = resp.Body.Close() + return awsError(http.StatusInternalServerError, "InternalServerError"), nil +} + +func TestDedupeDynamo_Conformance(t *testing.T) { + t.Parallel() + dedupetest.Run(t, func(t *testing.T) dedupetest.Harness { + table := newDynamoTable() + // failAfter < 0 is off; otherwise the put after that many fails once. + var failAfter, puts atomic.Int64 + failAfter.Store(-1) + fault := config.WithHTTPClient(&faultyHTTP{next: http.DefaultClient, fault: func(target string) (*http.Response, error, bool) { + if target != putItem || failAfter.Load() < 0 || puts.Add(1) <= failAfter.Load() { + return nil, nil, false + } + failAfter.Store(-1) + return awsError(http.StatusBadRequest, "ValidationException"), nil, true + }}) + d := dynamoClient(t, table, dedupe.DynamoConfig{}, fault) + require.NoError(t, d.CreateTable(t.Context())) + require.NoError(t, d.Check(t.Context())) + return dedupetest.Harness{ + Factory: d.Tenant, + Peer: dynamoClient(t, table, dedupe.DynamoConfig{}).Tenant, + FailNextReserve: func(n int) { + puts.Store(0) + failAfter.Store(int64(n)) + }, + } + }) +} + +// 32 clients — 32 pods — race one id: DynamoDB's condition, not anything in +// process, is what lets exactly one through. +func TestDedupeDynamo_ThirtyTwoClientsOneID(t *testing.T) { + t.Parallel() + table := newDynamoTable() + first := dynamoClient(t, table, dedupe.DynamoConfig{}) + require.NoError(t, first.CreateTable(t.Context())) + const n = 32 + stores := make([]*dedupe.Managed, n) + for i := range stores { + stores[i] = dynamoClient(t, table, dedupe.DynamoConfig{}).Tenant("acme") + require.NoError(t, stores[i].Apply(true)) + } + k := []dedupe.Key{{Table: "events", ID: "e1"}} + race := func() map[dedupe.Status][]dedupe.Claim { + got := make([]dedupe.Claim, n) + start := make(chan struct{}) + var wg sync.WaitGroup + for i, s := range stores { + wg.Go(func() { + <-start + c, err := s.Reserve(context.Background(), k, time.Minute) + if assert.NoError(t, err) { + got[i] = c[0] + } + }) + } + close(start) + wg.Wait() + by := map[dedupe.Status][]dedupe.Claim{} + for _, c := range got { + by[c.Status] = append(by[c.Status], c) + } + return by + } + by := race() + require.Len(t, by[dedupe.Claimed], 1, "exactly one client claims the id") + assert.Len(t, by[dedupe.InFlight], n-1) + require.NoError(t, stores[0].Commit(t.Context(), by[dedupe.Claimed], 0)) + assert.Len(t, race()[dedupe.Duplicate], n, "and every client then sees it committed") +} + +func TestDedupeDynamo_Throttled(t *testing.T) { + t.Parallel() + table := newDynamoTable() + require.NoError(t, dynamoClient(t, table, dedupe.DynamoConfig{}).CreateTable(t.Context())) + for _, code := range []string{"ThrottlingException", "ProvisionedThroughputExceededException", "RequestLimitExceeded"} { + t.Run(code, func(t *testing.T) { + t.Parallel() + var sent atomic.Int64 + d := dynamoClient(t, table, dedupe.DynamoConfig{MaxAttempts: 2}, config.WithHTTPClient(&faultyHTTP{ + next: http.DefaultClient, + fault: func(target string) (*http.Response, error, bool) { + if target != putItem { + return nil, nil, false + } + sent.Add(1) + return awsError(http.StatusBadRequest, code), nil, true + }, + })) + m := d.Tenant("acme") + require.NoError(t, m.Apply(true)) + _, err := m.Reserve(t.Context(), []dedupe.Key{{Table: "events", ID: "e1"}}, time.Minute) + require.ErrorIs(t, err, dedupe.ErrUnavailable, "a throttle is worth retrying") + assert.Equal(t, int64(2), sent.Load(), "the SDK retried it once first") + }) + } +} + +// The SDK's retry of an applied put fails its condition on the put's own +// item, which is still the caller's claim: without that, the id would be +// held InFlight for the lease by a claim nobody commits or releases. +func TestDedupeDynamo_RetriedPutKeepsItsClaim(t *testing.T) { + t.Parallel() + table := newDynamoTable() + lossy := &landThenFail{next: http.DefaultClient} + d := dynamoClient(t, table, dedupe.DynamoConfig{}, config.WithHTTPClient(lossy)) + require.NoError(t, d.CreateTable(t.Context())) + m := d.Tenant("acme") + require.NoError(t, m.Apply(true)) + peer := dynamoClient(t, table, dedupe.DynamoConfig{}).Tenant("acme") + require.NoError(t, peer.Apply(true)) + k := []dedupe.Key{{Table: "events", ID: "e1"}} + + claims, err := m.Reserve(t.Context(), k, time.Minute) + require.NoError(t, err) + require.True(t, lossy.failed.Load(), "the applied attempt was answered 500") + require.Equal(t, dedupe.Claimed, claims[0].Status) + other, err := peer.Reserve(t.Context(), k, time.Minute) + require.NoError(t, err) + assert.Equal(t, dedupe.InFlight, other[0].Status) + + require.NoError(t, m.Release(t.Context(), claims)) + other, err = peer.Reserve(t.Context(), k, time.Minute) + require.NoError(t, err) + assert.Equal(t, dedupe.Claimed, other[0].Status, "the claim's token was the applied put's, so Release freed the id") +} + +func TestDedupeDynamo_Unreachable(t *testing.T) { + t.Parallel() + d, err := dedupe.NewDynamo(t.Context(), dedupe.DynamoConfig{ + Table: "dedupe", Region: "us-east-1", Endpoint: "http://127.0.0.1:1", MaxAttempts: 1, + }, config.WithCredentialsProvider(credentials.NewStaticCredentialsProvider("local", "local", ""))) + require.NoError(t, err) + m := d.Tenant("acme") + require.NoError(t, m.Apply(true)) + k := []dedupe.Key{{Table: "events", ID: "e1"}} + for range 5 { + _, err = m.Reserve(t.Context(), k, time.Minute) + require.ErrorIs(t, err, dedupe.ErrUnavailable) + } + _, err = m.Reserve(t.Context(), k, time.Minute) + require.ErrorIs(t, err, dedupe.ErrUnavailable) + assert.Contains(t, err.Error(), "short-circuited", "five failures in a second open the breaker") + assert.ErrorIs(t, d.Check(t.Context()), dedupe.ErrUnavailable) +} + +func TestDedupeDynamo_ConfigErrorsAreNotUnavailable(t *testing.T) { + t.Parallel() + d := dynamoClient(t, "no_such_table", dedupe.DynamoConfig{}) + m := d.Tenant("acme") + require.NoError(t, m.Apply(true)) + _, err := m.Reserve(t.Context(), []dedupe.Key{{Table: "events", ID: "e1"}}, time.Minute) + require.Error(t, err) + var missing *types.ResourceNotFoundException + assert.ErrorAs(t, err, &missing) + assert.False(t, errors.Is(err, dedupe.ErrUnavailable), "a missing table is a config bug, not worth retrying") + require.Error(t, d.Check(t.Context())) +} + +func TestDedupeDynamo_Check(t *testing.T) { + t.Parallel() + raw := rawDynamo(t) + table := newDynamoTable() + _, err := raw.CreateTable(t.Context(), &dynamodb.CreateTableInput{ + TableName: aws.String(table), + BillingMode: types.BillingModePayPerRequest, + AttributeDefinitions: []types.AttributeDefinition{{AttributeName: aws.String("pk"), AttributeType: types.ScalarAttributeTypeB}}, + KeySchema: []types.KeySchemaElement{{AttributeName: aws.String("pk"), KeyType: types.KeyTypeHash}}, + }) + require.NoError(t, err) + assert.ErrorContains(t, dynamoClient(t, table, dedupe.DynamoConfig{}).Check(t.Context()), "must be a string") + + fresh := newDynamoTable() + d := dynamoClient(t, fresh, dedupe.DynamoConfig{}) + require.NoError(t, d.CreateTable(t.Context())) + require.NoError(t, d.CreateTable(t.Context()), "an existing table is left alone") + ttl, err := raw.DescribeTimeToLive(t.Context(), &dynamodb.DescribeTimeToLiveInput{TableName: aws.String(fresh)}) + require.NoError(t, err) + assert.Equal(t, "ex", aws.ToString(ttl.TimeToLiveDescription.AttributeName)) + assert.Equal(t, types.TimeToLiveStatusEnabled, ttl.TimeToLiveDescription.TimeToLiveStatus) +} + +// Expiry is the item's ex, in epoch seconds, and never depends on TTL having +// deleted the item. +func TestDedupeDynamo_Expiry(t *testing.T) { + t.Parallel() + raw := rawDynamo(t) + table := newDynamoTable() + d := dynamoClient(t, table, dedupe.DynamoConfig{}) + require.NoError(t, d.CreateTable(t.Context())) + m := d.Tenant("acme") + require.NoError(t, m.Apply(true)) + pk := func(id string) string { + return string(dedupe.AppendKey(nil, dedupe.KeyPrefix("acme"), dedupe.Key{Table: "events", ID: id})) + } + item := func(id string) map[string]types.AttributeValue { + out, err := raw.GetItem(t.Context(), &dynamodb.GetItemInput{ + TableName: aws.String(table), ConsistentRead: aws.Bool(true), + Key: map[string]types.AttributeValue{"pk": &types.AttributeValueMemberS{Value: pk(id)}}, + }) + require.NoError(t, err) + return out.Item + } + num := func(av types.AttributeValue) int64 { + n, err := strconv.ParseInt(av.(*types.AttributeValueMemberN).Value, 10, 64) + require.NoError(t, err) + return n + } + + before := time.Now() + claims, err := m.Reserve(t.Context(), []dedupe.Key{{Table: "events", ID: "kept"}, {Table: "events", ID: "brief"}}, 30*time.Second) + require.NoError(t, err) + pending := item("kept") + assert.Equal(t, "1", pending["st"].(*types.AttributeValueMemberN).Value) + assert.InDelta(t, before.Add(30*time.Second).Unix(), num(pending["ex"]), 2, "a pending item's ex is its lease end") + + require.NoError(t, m.Commit(t.Context(), claims[:1], 0)) + require.NoError(t, m.Commit(t.Context(), claims[1:], time.Hour)) + assert.NotContains(t, item("kept"), "ex", "retention 0 writes no ex, so TTL never takes it") + assert.InDelta(t, before.Add(time.Hour).Unix(), num(item("brief")["ex"]), 2, "a commit's ex is its retention end") + + // TTL deletes lazily; an item whose ex has passed is absent all the same. + for _, st := range []string{"1", "2"} { + _, err = raw.PutItem(t.Context(), &dynamodb.PutItemInput{TableName: aws.String(table), Item: map[string]types.AttributeValue{ + "pk": &types.AttributeValueMemberS{Value: pk("stale-" + st)}, + "st": &types.AttributeValueMemberN{Value: st}, + "ex": &types.AttributeValueMemberN{Value: strconv.FormatInt(time.Now().Add(-time.Minute).Unix(), 10)}, + "tk": &types.AttributeValueMemberB{Value: []byte("old")}, + }}) + require.NoError(t, err) + } + got, err := m.Reserve(t.Context(), []dedupe.Key{{Table: "events", ID: "stale-1"}, {Table: "events", ID: "stale-2"}, {Table: "events", ID: "kept"}}, time.Minute) + require.NoError(t, err) + assert.Equal(t, []dedupe.Status{dedupe.Claimed, dedupe.Claimed, dedupe.Duplicate}, + []dedupe.Status{got[0].Status, got[1].Status, got[2].Status}) +} + +func TestDedupeDynamo_CreateTableNeedsEndpoint(t *testing.T) { + t.Parallel() + d, err := dedupe.NewDynamo(t.Context(), dedupe.DynamoConfig{Table: "dedupe", Region: "us-east-1"}, + config.WithCredentialsProvider(credentials.NewStaticCredentialsProvider("local", "local", ""))) + require.NoError(t, err) + require.ErrorIs(t, d.CreateTable(t.Context()), dedupe.ErrCreateTableNeedsEndpoint) +} diff --git a/tests/integration/ingest_outage_test.go b/tests/integration/ingest_outage_test.go index 752d84b21..53faa2d0b 100644 --- a/tests/integration/ingest_outage_test.go +++ b/tests/integration/ingest_outage_test.go @@ -18,6 +18,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" ) // TestIngest_ClickHouseOutage_RetriedNotDeadLettered stops a real ClickHouse @@ -39,7 +40,7 @@ func TestIngest_ClickHouseOutage_RetriedNotDeadLettered(t *testing.T) { const table = "outage_events" require.NoError(t, ch.conn.Exec(ctx, "CREATE TABLE "+table+" (id UInt32) ENGINE = MergeTree ORDER BY id")) - broker, err := mq.NewEmbedded(t.TempDir()) + broker, err := mq.NewEmbedded(storedir.New(t)) require.NoError(t, err) t.Cleanup(func() { _ = broker.Close() }) require.NoError(t, broker.SetMaxBytes(ctx, tenant.Default, 64<<20)) diff --git a/tests/integration/query_errors_test.go b/tests/integration/query_errors_test.go index 9eccd851d..d29660efa 100644 --- a/tests/integration/query_errors_test.go +++ b/tests/integration/query_errors_test.go @@ -22,6 +22,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/app" "github.com/Wave-RF/WaveHouse/internal/chconn" "github.com/Wave-RF/WaveHouse/internal/config" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" ) // queryError is the error envelope a failed ClickHouse query answers with. @@ -115,7 +116,7 @@ func TestQueryErrors_ClickHouseDown(t *testing.T) { ln, err := lc.Listen(ctx, "tcp", "127.0.0.1:0") require.NoError(t, err) cfg := &config.Config{ - DataDir: t.TempDir(), + DataDir: storedir.New(t), Server: config.Server{ShutdownTimeout: 10}, ClickHouse: config.ClickHouse{Password: testCHPassword}, MQ: config.MQ{Backend: config.MQEmbedded}, diff --git a/tests/integration/setup_test.go b/tests/integration/setup_test.go index 477bddd83..240417b69 100644 --- a/tests/integration/setup_test.go +++ b/tests/integration/setup_test.go @@ -51,6 +51,9 @@ type testEnv struct { embeddedMQ mq.Broker baseURL string // the wired API server, e.g. http://127.0.0.1:41234 registry *discovery.SchemaRegistry + // dynamoEndpoint is dynamodb-local, for the DynamoDB dedupe backend's + // tests; the wired app does not use it. + dynamoEndpoint string } var sharedEnv *testEnv @@ -137,6 +140,13 @@ func setup() (int, func()) { _ = ch.container.Terminate(context.Background()) }) + ddb, endpoint, err := startDynamoDBLocal(ctx) + if err != nil { + fmt.Fprintf(os.Stderr, "integration setup: dynamodb-local: %v\n", err) + return 1, cleanup + } + cleanups.push(func() { _ = ddb.Terminate(context.Background()) }) + settingsDir, err := writeTestSettings(ch) if err != nil { fmt.Fprintf(os.Stderr, "integration setup: settings: %v\n", err) @@ -203,6 +213,8 @@ func setup() (int, func()) { embeddedMQ: a.MQ(), baseURL: baseURL, registry: a.Registry(), + + dynamoEndpoint: endpoint, } return 0, cleanup } @@ -387,6 +399,28 @@ func startClickHouse(ctx context.Context) (*chInstance, error) { return ch, nil } +// startDynamoDBLocal starts dynamodb-local in memory (no volume) and +// returns it with its endpoint URL. +func startDynamoDBLocal(ctx context.Context) (testcontainers.Container, string, error) { + container, err := testcontainers.GenericContainer(ctx, testcontainers.GenericContainerRequest{ + ContainerRequest: testcontainers.ContainerRequest{ + Image: "amazon/dynamodb-local:3.3.1", + Cmd: []string{"-jar", "DynamoDBLocal.jar", "-inMemory"}, + ExposedPorts: []string{"8000/tcp"}, + WaitingFor: wait.ForListeningPort("8000/tcp").WithStartupTimeout(60 * time.Second), + }, + Started: true, + }) + if err != nil { + return nil, "", fmt.Errorf("start container: %w", err) + } + endpoint, err := container.PortEndpoint(ctx, "8000/tcp", "http") + if err != nil { + return container, "", fmt.Errorf("endpoint: %w", err) + } + return container, endpoint, nil +} + func waitForNativeReady(ctx context.Context, conn driver.Conn, timeout time.Duration) error { pingCtx, cancel := context.WithTimeout(ctx, timeout) defer cancel() diff --git a/tests/integration/tenants_test.go b/tests/integration/tenants_test.go index 16d888ba5..e3c756b67 100644 --- a/tests/integration/tenants_test.go +++ b/tests/integration/tenants_test.go @@ -20,6 +20,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/app" "github.com/Wave-RF/WaveHouse/internal/config" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" ) // TestNestedDirectory_PerTenantPoolsAndDiscovery boots the real wiring over @@ -58,7 +59,7 @@ func TestNestedDirectory_PerTenantPoolsAndDiscovery(t *testing.T) { ln, err := lc.Listen(ctx, "tcp", "127.0.0.1:0") require.NoError(t, err) cfg := &config.Config{ - DataDir: t.TempDir(), + DataDir: storedir.New(t), Server: config.Server{ShutdownTimeout: 10}, ClickHouse: config.ClickHouse{Password: testCHPassword}, Auth: config.Auth{OperatorKey: operatorKey},