From b83e67075a85017e31ae7d31fabd1b1bf939d35e Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Fri, 25 Sep 2026 03:49:16 -0400 Subject: [PATCH 01/11] feat(dedupe): retention per tenant and table, and an expiry sweep dedupe.retention (required, "0" = forever) and its per-table override are read per record and passed to Commit. Validation refuses a finite retention below the queue's two-minute duplicate window. The embedded Pebble store deletes expired keys and the version-0 keys in an hourly background sweep, counted by wavehouse_dedupe_swept_keys_total. Part of #613. Refs #220. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01EJr5tY4WQUy2sc4MbW67vL --- AGENTS.md | 4 +- CHANGELOG.md | 3 +- cmd/wavehouse/validate_test.go | 2 +- config.yaml | 2 +- deployments/compose/settings/config.json | 1 + docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/deployment.md | 6 +- docs/src/content/docs/durability.md | 4 +- docs/src/content/docs/settings-directory.mdx | 9 +- internal/api/ingest.go | 66 +++++--- internal/api/ingest_retention_test.go | 104 ++++++++++++ internal/api/ingest_test.go | 40 +++-- internal/api/ingest_window_test.go | 4 +- internal/api/settings_test.go | 2 +- internal/app/app_test.go | 14 +- internal/dedupe/embedded.go | 65 ++++++-- internal/dedupe/sweep.go | 151 ++++++++++++++++++ internal/dedupe/sweep_test.go | 158 +++++++++++++++++++ internal/settings/registry_test.go | 4 +- internal/settings/seed/config.json | 1 + internal/settings/settings.go | 18 ++- internal/settings/store.go | 31 +++- internal/settings/store_test.go | 21 ++- internal/settings/validate.go | 30 +++- internal/settings/validate_test.go | 39 ++++- internal/testutil/mocks.go | 13 +- tests/e2e/fixtures/settings/config.json | 1 + 27 files changed, 693 insertions(+), 102 deletions(-) create mode 100644 internal/api/ingest_retention_test.go create mode 100644 internal/dedupe/sweep.go create mode 100644 internal/dedupe/sweep_test.go diff --git a/AGENTS.md b/AGENTS.md index ffbd3c16f..b26db3e3a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -35,7 +35,7 @@ Eighteen internal packages under `internal/` (plus `internal/testutil/` for shar - **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (nil for a tenant on no pool, a `503`), `Target` (the tenant's own HTTP wiring over its pool's TLS config), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config - **`chsql/`** — dependency-free ClickHouse SQL helpers shared by `query`/`policy` (avoids an import cycle): `QuoteIdent` (backtick-quote every identifier) + `BindUnsafe` (reject names with a literal `?`) - **`config/`** — YAML + env var config loading (cleanenv); strict on both sides (undeclared YAML key, unbound `WH_*` variable) and probes `data_dir` writability — boot is the validator, there is no dry run -- **`dedupe/`** — `Deduplicator` interface (two-phase `Reserve`/`Commit`/`Release` over `Key{Table, ID}`; every backend passes the `dedupetest` conformance suite) → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant and table, pending claims in memory, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) +- **`dedupe/`** — `Deduplicator` interface (two-phase `Reserve`/`Commit`/`Release` over `Key{Table, ID}`; every backend passes the `dedupetest` conformance suite) → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant and table, pending claims in memory, committed ids stored with their expiry and deleted by an hourly background sweep along with the version-0 keys from before the table joined the key, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) - **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version, + `Validate()` for ingest payloads + `CanonicalizeTimestamps()` rewriting top-level `DateTime`/`DateTime64` column values to the canonical RFC 3339 UTC wire form pre-publish (Key Design Decision #19) - **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else — `row` is one positional `JSONCompactEachRow` line and `columns` names its slots, the table's **insertable** columns (a `MATERIALIZED`/`ALIAS` column cannot be named in an `INSERT`); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs. In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`), so non-admin callers never reach the proxy. A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) - **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20). and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal; `WithIdempotencyKey` makes a republish inside the queue's duplicate window a no-op), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, token encoding, stream names, sequences, and ack floors are private to the one implementation, `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`); `internal/app` constructs it and hands everything else a `mq.Broker` @@ -58,7 +58,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 5. **Per-tenant-table batching** — the worker groups events by tenant table (the tenant read off each message's `mq.Topic`), so one INSERT never mixes tenants and a batch invalidates its own tenant's cache namespaces; then it splits each batch by column list (`groupByColumns`), emitting one `INSERT INTO … (cols) FORMAT JSONCompactEachRow` per distinct list so a schema change mid-stream can't corrupt a statement. Each tenant table's batch is independent. 6. **Dead Letter Queue** — failed batch inserts publish to the tenant's own dead-letter queue (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format`, or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. 7. **Auth: always on, fail-loud, decoupled from authz (security)** — the JWT middleware always runs (no `auth.enabled`/`dev_mode` flag); it verifies with HMAC **or** JWKS (not both), with accepted `alg` pinned to the active verifier and checked before any key is used (rejects `alg:none` and cross-family confusion). No/invalid/expired token → empty role → policy `default_role`, with the bad-token reason stashed so a denying gate returns a loud `401`, not a bare `403`; the one token outcome that never reaches `default_role` is a verifier still fetching its JWKS (`auth.ErrVerifierPending` → `503` + `Retry-After`, `api.refuseUnverifiable`). Elevated access needs a valid granted role. **Sanctioned exception:** a configured non-JWT operator key (`auth.operator_key`; presented via `Authorization: Operator ` or the `X-Operator-Key` alias) deliberately couples authN+authZ — a constant-time match authorizes a full-access platform operator (stamps the admin role plus an operator bit) independent of the verifier (see #11). Detail: architecture.md § `api/` + `internal/auth`; see also #11, §Security Considerations. -8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant and table; claims are two-phase, one call per phase per window of up to 256 records — `Reserve` → publish (under the id's idempotency key) → `Commit`, or `Release` when the publish definitely failed, while one whose outcome is unknown is left to lapse; a store that cannot answer is a `503` + `Retry-After`; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key, overridable per table. +8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant and table; claims are two-phase, one call per phase per window of up to 256 records — `Reserve` → publish (under the id's idempotency key) → `Commit`, or `Release` when the publish definitely failed, while one whose outcome is unknown is left to lapse; a store that cannot answer is a `503` + `Retry-After`; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key and `dedupe.retention` how long a committed id stays a duplicate (`"0"` = forever, else at least the queue's two-minute duplicate window), both overridable per table. 9. **Singleflight** — the cached read handlers coalesce concurrent misses (`x/sync/singleflight`) under the tenant-led cache key to prevent cache stampede, per tenant. 10. **Active Sweeper** — purges NATS messages that are both ACKed (written to CH) and older than the gap window; SSE gap-fill uses `DeliverByStartTime`, no in-process ring buffer. 11. **Hasura-style access control: fail-closed (security)** — `policy.IsAdmin` (role == `admin_role`, **exact case-sensitive**, default `"admin"`) is the single admin check, shared by `Evaluate`/`ResolveRole`/`Validate`/the `/v1/ops` gate/`RoleAllowed`. Empty/absent role matches nothing (no `"*"` wildcard); `Validate` rejects empty role keys; a `nil` policy (deleted) denies **everyone incl. admin** via a role — a total lockout for token-based callers, so recovery is writing `policies.json` and reloading, never an implicit admin grant (**exception:** the operator key's `auth.IsOperator` bit passes the `/v1/ops` gate even under a `nil` policy — a deliberate break-glass that can `POST /v1/ops/settings/reload` over HTTP, see #7). Over a nested settings directory the `/v1/ops` gate reads no policy at all — those routes reach every tenant, so the operator key alone passes and an admin-role token gets `403`; `api.NewRouter` decides that from the registry's shape, not from what was wired. `default_role` is the one sanctioned roleless exception (`ResolveRole` maps empty → it pre-eval); `default_role == admin_role` is permitted but dev-only and loudly warned (`policy.DefaultRoleGrantsAdmin`). Preserve when touching `internal/policy` (policy twin of #13; see #159). Detail: architecture.md § `policy/`. diff --git a/CHANGELOG.md b/CHANGELOG.md index 878072338..f47f81ba6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Settings-directory validation — `wavehouse validate [dir]`** (`internal/settings/` (new: `settings.go`, `validate.go`, `decode.go`, `finding.go`, + tests), `cmd/wavehouse/validate.go` (new, + tests), `cmd/wavehouse/main.go`): first piece of the file-based control plane (settings live in a directory of JSON documents — `roles.json`, `policies.json`, `pipes.json`, `config.json` — that a running instance will hot-reload; this change is validation-only — boot loading and reload wiring land separately). `settings.Validate(dir)` is the single gate every consumer of the directory runs: deliberately pure (no network, no ClickHouse — table/column existence stays with schema discovery, per Bring-Your-Own-Schema), and it collects **all** findings in one pass instead of failing on the first. Checks, layered: the directory holds exactly the four files (a missing file is an error — an empty document is `{}`, so absence always means deletion or a wrong path; any unexpected entry — file or directory — is an error so a typoed `polices.json` or a stray backup can't be silently ignored; dot-prefixed entries are the one carve-out, since erroring on vim swap files or the `..data` machinery Kubernetes ConfigMap mounts publish through would break hand editing and the cloud fan-out's mount pattern alike); strict JSON syntax (unknown fields rejected — the JSON form of the retired-config-key trap; empty/truncated files rejected, never read as an empty document; a leading UTF-8 byte order mark named as such instead of surfacing as a cryptic invalid-character error; a directory, unreadable file, or non-regular file (a FIFO would hang the read forever waiting for a writer; a stat gate rejects it — following symlinks, so Kubernetes ConfigMap mounts' symlink layout still passes) squatting on a settings filename named as the one real problem, not double-reported as "missing"; a top-level `null` rejected — the one well-formed document that decodes into a zero value without error, so it would silently read as "no settings"; trailing content rejected; duplicated object keys detected by a token-level pass, since `encoding/json` silently keeps the last copy); per-file shape rules (role names non-empty/unique, pipe names/SQL/param types, `config.json` bounds mirroring boot-config validation — its sections are the *tenant-owned* behavioral tunables (dedupe id_field/require_id plus per-table overrides under `dedupe.tables` — each entry overrides only the fields it names, resolving table → global → compiled default per field, so the effective id_field can never be empty — an explicit empty, whitespace-only, or whitespace-padded id_field is rejected at both levels, since an exact-match JSON key lookup would silently miss every row ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s shape, unblocked by the file design since table names are runtime-resolved like policy grants); query default_max_rows, schema refresh_interval, CORS origins); platform-owned knobs like the SSE keepalives deliberately stay boot config); and cross-file referential integrity (every role a policy grant, `default_role`/`admin_role`, or pipe allowlist references must be declared in `roles.json`; an empty role string in a grant or allowlist is named as such — it matches no request and authorizes nobody). Warnings don't invalidate: a grant scoping the admin role (an unconditional bypass — dead config), `default_role` = admin, and a `default` on a required pipe parameter are flagged but legal. An empty `policies.json` means no policy — fail closed, matching deleted-policy semantics — and draws a warning naming the total lockout, so it announces itself at validation time instead of one 403 at a time. The CLI (`cmd/wavehouse/validate.go`, following the `health` subcommand pattern) takes the directory as an argument or from `WH_SETTINGS_DIR`, prints findings, and exits 0/1/2 (valid/invalid/usage) so CI and operators can gate config changes before they reach a running instance. The dispatch in `main.go` also grows `help` and `version` subcommands, and an unknown command is now a usage error instead of silently falling through and starting the server (`wavehouse validat` booting a listener is not a typo anyone wants); each subcommand parses its arguments with a stdlib `flag.FlagSet`, so `wavehouse -h` prints command-specific help and a stray flag or argument is a usage error rather than being silently swallowed. `WH_SETTINGS_DIR` has a single authority: `config.EnvSettingsDir`, with a reflection test pinning the `settings.dir` struct tag to it. The directory's location joins boot config as `settings.dir` (`WH_SETTINGS_DIR`; `internal/config/config.go`, `config.yaml`, `docs/src/content/docs/configuration.mdx`) — boot-tier by necessity, since it's the pointer the reload machinery follows; no default, same silent-misconfiguration reasoning as `policy.file_path`. - **Docs-site analytics for search, code copies, 404s, docs section, and live-demo connectivity** (`docs/src/components/DocsTracking.astro` (new), `docs/src/components/{PostHog,Footer,LiveDemo}.astro`): the site tracked its own CTAs but nothing a reader did on the way to one, so the questions that decide what to write next — what people search for and *don't* find, which snippets get copied, which dead links keep getting followed — had no data behind them. `docs_search` fires a second after the query settles rather than once per keystroke, carrying `query` and `result_count` read off Pagefind's own results message (the rendered list is capped at its page size, so counting the DOM would under-report); `result_count: 0` is the event worth having. `code_copied` (`page`, `language`) watches Expressive Code's copy buttons from the document rather than re-binding every code block on every navigation — the hero's install chip is not an EC block and keeps its own `hero_install_copied`. `docs_404` (`path`, `referrer`) turns broken inbound links into a list instead of a hunch. A `doc_section` property (the first path segment, `home` for `/`) puts every event in a docs area without each tracker carrying its own copy; it's stamped at capture time by a `before_send` hook in `posthog.init()` rather than `register()`, because a queued `register()` replays only after init has already captured the first hard-load `$pageview` — which would then carry the previous visit's persisted value — and `history_change` navigations update the URL before capture fires, so reading `location` in the hook is always current. `live_demo_connected` fires once per mount when the hero's SSE feed comes up rather than on its first row — named for what it measures (the demo backend answered), since a quiet minute on the repo is not a disengaged reader. The three site-wide trackers share one new `DocsTracking.astro` rendered from the footer (like `MermaidZoom` / `ScrollHints`) and delegate from `document`, since Pagefind, Expressive Code, and the 404 route all own their own markup — some of it created after page load. +- **Dedupe retention per tenant and table, and a sweep that deletes expired ids** (`internal/settings/{settings,validate,store}.go` (+ tests), `internal/settings/seed/config.json`, `internal/dedupe/{embedded,sweep}.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/testutil/mocks.go`, `deployments/compose/settings/config.json`, `tests/e2e/fixtures/settings/config.json`, `config.yaml`, `docs/src/content/docs/{settings-directory.mdx,deployment,durability,architecture}.md`, `AGENTS.md`): [#220](https://github.com/Wave-RF/WaveHouse/issues/220). `config.json` gains **`dedupe.retention`, a required key**, overridable per table in `dedupe.tables.
.retention`: how long a committed id stays a duplicate, as a duration string (`"720h"`), or `"0"` to keep it forever, which is the seed value and the behaviour before this release. **Every existing `config.json` must add it**; `"retention": "0"` changes nothing. A finite retention below `"2m"`, the ingest queue's duplicate window, is refused rather than raised to the minimum: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and dropped by the queue as a copy while the client was told it was accepted. It is hot-reloadable and read per record like `id_field`; a change applies to ids committed after it, and a reload landing mid-window commits each record with the retention it was prepared under. The embedded Pebble store already treated an expired id as new; it now also deletes expired keys, and the version-0 keys the key-layout change left behind, in a background sweep that starts a minute after the instance opens and repeats hourly. It deletes 1,024 keys per chunk without fsync, under a lock `Commit` also takes, so an id committed again after the sweep read it is never deleted. New metric `wavehouse_dedupe_swept_keys_total{reason="expired"|"version_0"}`. `settings.Store.DedupeFor` now returns a `settings.Dedupe` struct rather than three values. ### Changed @@ -78,7 +79,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Fixed -- **Dedupe claims an id, publishes, then commits it — and keys it by tenant, table and id** (`internal/dedupe/{dedupe,key,embedded,managed}.go` (+ tests), `internal/dedupe/dedupetest/` (new), `internal/api/ingest.go` (+ tests), `internal/settings/validate.go` (+ tests), `internal/testutil/mocks.go`, `internal/app/app_test.go`, `AGENTS.md`, `docs/src/content/docs/{architecture,api,deployment}.md`, `settings-directory.mdx`, `sdk/reference.md`): `CheckAndMark` is replaced by a two-phase `Reserve` → `Commit` / `Release` contract with a lease on the pending claim, and every backend now runs one conformance suite. Four bugs go with it. Two concurrent requests carrying one id no longer both publish it: Pebble's check and claim happen under one lock, and the loser answers `503` with `Retry-After` while the winner is still publishing ([#390](https://github.com/Wave-RF/WaveHouse/issues/390)). A publish that fails gives its id back, so the retry a `503` asks for is published instead of skipped as a duplicate of a record that never reached the queue ([#384](https://github.com/Wave-RF/WaveHouse/issues/384)). The same id in two tables is two ids ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s keyspace half). An explicit `null` id is a missing id — rejected under `require_id`, published un-deduped otherwise — instead of the one id `""` that made every null record after the first a duplicate ([#370](https://github.com/Wave-RF/WaveHouse/issues/370)). **Upgrade:** the key layout changes, so an id seen before the upgrade is accepted once more after it; nothing is migrated, and the old keys are left in `/pebble`, unread, with nothing removing them yet ([#220](https://github.com/Wave-RF/WaveHouse/issues/220) tracks the sweep) ([Deployment → Upgrading across the dedupe key change](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md)). A table name holding a NUL byte is refused in `dedupe.tables` and `dlq.tables`. New metrics: `wavehouse_dedupe_commit_failed_total` (a published record whose id failed to commit; the claim lapses with its lease) and `wavehouse_dedupe_hashed_id_total` (an id over 1,024 bytes, stored as its SHA-256). +- **Dedupe claims an id, publishes, then commits it — and keys it by tenant, table and id** (`internal/dedupe/{dedupe,key,embedded,managed}.go` (+ tests), `internal/dedupe/dedupetest/` (new), `internal/api/ingest.go` (+ tests), `internal/settings/validate.go` (+ tests), `internal/testutil/mocks.go`, `internal/app/app_test.go`, `AGENTS.md`, `docs/src/content/docs/{architecture,api,deployment}.md`, `settings-directory.mdx`, `sdk/reference.md`): `CheckAndMark` is replaced by a two-phase `Reserve` → `Commit` / `Release` contract with a lease on the pending claim, and every backend now runs one conformance suite. Four bugs go with it. Two concurrent requests carrying one id no longer both publish it: Pebble's check and claim happen under one lock, and the loser answers `503` with `Retry-After` while the winner is still publishing ([#390](https://github.com/Wave-RF/WaveHouse/issues/390)). A publish that fails gives its id back, so the retry a `503` asks for is published instead of skipped as a duplicate of a record that never reached the queue ([#384](https://github.com/Wave-RF/WaveHouse/issues/384)). The same id in two tables is two ids ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s keyspace half). An explicit `null` id is a missing id — rejected under `require_id`, published un-deduped otherwise — instead of the one id `""` that made every null record after the first a duplicate ([#370](https://github.com/Wave-RF/WaveHouse/issues/370)). **Upgrade:** the key layout changes, so an id seen before the upgrade is accepted once more after it; nothing is migrated, and the old keys are left in `/pebble`, unread, deleted by the retention sweep below ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)) ([Deployment → Upgrading across the dedupe key change](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md)). A table name holding a NUL byte is refused in `dedupe.tables` and `dlq.tables`. New metrics: `wavehouse_dedupe_commit_failed_total` (a published record whose id failed to commit; the claim lapses with its lease) and `wavehouse_dedupe_hashed_id_total` (an id over 1,024 bytes, stored as its SHA-256). - **Ingest runs in windows of 256 records over the dedupe contract, and a dedupe store that cannot answer is a `503`** (`internal/api/ingest.go` (+ tests), `internal/mq/{mq,embedded}.go` (+ tests), `internal/dedupe/key.go` (+ tests), `internal/testutil/mocks.go`, `AGENTS.md`, `docs/src/content/docs/{api,architecture,durability}.md`, `settings-directory.mdx`, `sdk/reference.md`): each window of a request is prepared, then reserved in one dedupe call, published in order, and committed in one call, so a batch costs one dedupe round trip per phase per window rather than per record — on Pebble, one commit `fsync` per window (a 1,000-record batch: four syncs instead of a thousand, 24 ms against 5.7 s of dedupe time measured with the queue stubbed). Every deduped record is published under an idempotency key (`mq.WithIdempotencyKey`, JetStream's message id, derived by `dedupe.IdempotencyKey`), and each tenant's ingest stream now keeps an explicit two-minute duplicate window, which the 30-second lease fits inside. That closes the last path of [#384](https://github.com/Wave-RF/WaveHouse/issues/384): a publish that fails with an unknown outcome (anything but a full queue) keeps its record's claim until the lease lapses instead of releasing it, and a retry after the lease but within two minutes of the first publish is dropped by the queue if the first copy was stored (a later one is stored again). A dedupe store that is not open or that reports `dedupe.ErrUnavailable` now answers `503 {"error":"dedupe store unavailable"}` with `Retry-After: 5`, which the SDK retries, rather than `500 dedupe failed`. A mid-body read error or dedupe failure now drops the open window unpublished, where records before it used to be published; `wavehouse_dedupe_commit_failed_total` and `wavehouse_ingest_dedupe_disabled_total` count records, as before, now added a window at a time. - **An insert invalidates a table's cached results under every tenant the directory holds** (`internal/app/wire.go` (+ tests), `internal/settings/registry.go` (+ tests), `AGENTS.md`, `docs/src/content/docs/{deployment,architecture,ingest-pipeline}.md`): until [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6 gives each tenant its own ClickHouse, every tenant reads the same tables, but the ingest worker — which writes every event as tenant `0`'s until story 5 — bumped only tenant `0`'s cache namespaces after an insert, so another tenant's cached query could answer stale rows for up to its TTL (an hour at most). The cache the worker invalidates through now fans each bumped namespace out to every tenant the registry knows (the new `Registry.Known`), the named one and a rejected one included — a rejected tenant comes back into service with the entries it has, so leaving it out would let a folder repaired inside a TTL serve pre-insert rows; reads are untouched, so a tenant is still never served another's cached rows. The residual, a folder removed and restored inside a TTL, is closed since #610: a tenant back on a pool after an absence has its structured-query results orphaned at once (`Cache.InvalidateTenant`). Raised by CodeRabbit on #602. - **The `?token=` strip no longer repairs a query string that does not parse** (`internal/auth/auth.go`): `bearerToken` removed a query-string token by parsing the query, deleting `token`, and re-encoding what was left — and `url.ParseQuery` skips a pair it cannot read, so the re-encoding erased that pair. A handler that parses the query strictly in order to refuse a malformed one would then see a clean query: `GET /v1/ops/pipes?tenant=acme;x=1&token=…` would have answered `200` with the default tenant's pipes. The token is read exactly as before and a query that parses is rewritten exactly as before; a query that does not parse now loses its token pairs and nothing else, byte for byte. Pinned through `api.NewRouter` with the real authenticator, since a handler-level test never runs the middleware that rewrote the URL. diff --git a/cmd/wavehouse/validate_test.go b/cmd/wavehouse/validate_test.go index e2e18e6c0..5598cf650 100644 --- a/cmd/wavehouse/validate_test.go +++ b/cmd/wavehouse/validate_test.go @@ -19,7 +19,7 @@ func writeSettingsDir(t *testing.T, policies string) string { "roles.json": `{"roles": ["public"]}`, "policies.json": policies, "pipes.json": `{}`, - "config.json": `{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": 10000, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, + "config.json": `{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": 10000, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, } for name, content := range files { require.NoError(t, os.WriteFile(filepath.Join(dir, name), []byte(content), 0o600)) diff --git a/config.yaml b/config.yaml index 53a435029..358124171 100644 --- a/config.yaml +++ b/config.yaml @@ -63,7 +63,7 @@ auth: # clickhouse wiring (addr, http_port, http_scheme, database, username, # query_timeout, tls, headers, max_open_conns, max_idle_conns), auth # (jwks_url, role_claim), dedupe (enabled/id_field/ -# require_id + per-table overrides), dlq.enabled (+ per table), +# require_id/retention + per-table overrides), dlq.enabled (+ per table), # query.default_max_rows / timestamp_bucket_seconds, # schema.refresh_interval, stream keepalive_interval / keepalive_buckets / # gap_window_minutes, mq.max_bytes_gb, cors.allowed_origins — and every key diff --git a/deployments/compose/settings/config.json b/deployments/compose/settings/config.json index 030d76cc1..6b33f55c5 100644 --- a/deployments/compose/settings/config.json +++ b/deployments/compose/settings/config.json @@ -26,6 +26,7 @@ "enabled": false, "id_field": "event_id", "require_id": false, + "retention": "0", "tables": {} }, "dlq": { diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index e0812d7d5..29b6b2842 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -124,7 +124,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ - **dedupe.go** — the `Deduplicator` contract, two-phase: `Reserve(ctx, keys, lease)` answers one `Claim` per `Key{Table, ID}`, in order — `Claimed` (first sighting: the caller now holds a pending claim), `Duplicate` (committed earlier, or repeated earlier in the same call) or `InFlight` (another request holds a live claim) — and is atomic per key across every process sharing the backend; `Commit(ctx, claims, retention)` makes the published ids duplicates (retention `0` = forever); `Release(ctx, claims)` gives back ids whose records were definitely not published (a refused or never-sent publish; one whose outcome is unknown is left to lapse instead). A claim neither committed nor released lapses after its lease, so a request that dies mid-publish never strands an id. There is deliberately no read-only check: a separate read is how [#390](https://github.com/Wave-RF/WaveHouse/issues/390) happened. - **key.go** — the key layout every backend stores: a version byte, the tenant, a NUL, the table, a NUL, the id ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)). No tenant id holds a NUL and settings validation refuses one in a table name, so neither tenants nor tables share ids; an id over 1,024 bytes (or one starting `0xFF`) is stored as `0xFF` plus its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. -- **embedded.go** — `Embedded`, the [Pebble](https://github.com/cockroachdb/pebble) (embedded key-value store) implementation: every tenant's seen ids in one instance at `data_dir/pebble` ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 3). Pending claims live in memory beside it, in 64 locked shards: one process owns the instance, so a crash forgetting them is every lease lapsing at once, and the shard lock makes check-and-claim atomic. `Commit` writes every claim it is given in one batch and one fsync. `NewEmbedded(dataDir)` opens nothing; `Tenant(id)` is the `Factory` a `Stores` takes, building the tenant's `Managed` over its share of the instance, which opens with the first tenant store switched on and closes with the last one switched off. `Stats` reports the instance's figures for the system gauges, nil while it is closed. Pebble is per process: two pods on it do not share seen ids. +- **embedded.go** — `Embedded`, the [Pebble](https://github.com/cockroachdb/pebble) (embedded key-value store) implementation: every tenant's seen ids in one instance at `data_dir/pebble` ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 3). Pending claims live in memory beside it, in 64 locked shards: one process owns the instance, so a crash forgetting them is every lease lapsing at once, and the shard lock makes check-and-claim atomic. `Commit` writes every claim it is given in one batch and one fsync, each value carrying its expiry (`0` = never), which `Reserve` honors on read. A background sweep (`sweep.go`), started when the instance opens and stopped before it closes, deletes expired keys and the version-0 keys from before the table joined the key ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)): a minute after opening, then hourly, 1,024 keys per chunk, holding a lock `Commit` also takes, so a key re-committed after the sweep read it is never deleted; `wavehouse_dedupe_swept_keys_total{reason}` counts what it deletes. `NewEmbedded(dataDir)` opens nothing; `Tenant(id)` is the `Factory` a `Stores` takes, building the tenant's `Managed` over its share of the instance, which opens with the first tenant store switched on and closes with the last one switched off. `Stats` reports the instance's figures for the system gauges, nil while it is closed. Pebble is per process: two pods on it do not share seen ids. - **managed.go** — `Managed` wraps one store — opened through the function `NewManaged` takes, so the switch semantics are the same for every backend — behind the hot-reloadable `dedupe.enabled` switch: `Apply(enabled)` opens or closes it, idempotently, and in-flight calls are serialized against the swap, so flipping the key is a reload, not a restart. Every call returns `ErrDisabled` while switched off (the ingest handler publishes un-deduped and counts it — a reload-window race, not a mode) and `ErrUnavailable` while switched on but not open (ingest fails closed). `Reserve` also refuses an unstorable key, reads a lease `<= 0` as `DefaultLease`, and collapses a key repeated in one call before the backend sees it, once for every backend — so a backend may assume distinct keys, a positive lease, and only `Claimed` claims in `Commit` and `Release`. - **dedupetest/** — the conformance suite every backend runs: `Run(t, newHarness)` drives the contract above through a backend's `Factory` (claim, commit, release, lease lapse, retention, one claim among concurrent reserves from two clients, keyspaces, input order, late commit, stale release, failure mid-call); `Harness` optionally injects a clock and a mid-call failure. `Mark` is the old check-and-mark in one call, for tests that only need an id seen. - **stores.go** — `Stores` is one `Managed` per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7), built on first use through a `Factory` (`func(tenant.ID) *Managed`) — whether tenants share a backend is the factory's business (`Embedded.Tenant` puts them all in one Pebble instance), with nothing that holds the `Stores` changing. `For(id)` returns a tenant's store, built closed so a tenant adopted a moment ago answers `ErrDisabled` rather than having no store; `Retain(keep)` closes and forgets the stores of tenants no longer served, touching nothing on disk; `Close()` closes every store. `internal/app` drives it from the registry's `AfterAdopt` hook. diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index bd47e10a8..86347f444 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -169,7 +169,7 @@ WH_SETTINGS_DIR=/etc/wavehouse/settings WaveHouse keeps all embedded state under a single configurable root, `WH_DATA_DIR` (yaml: `data_dir`). Subdirectories are convention, not config: - `/nats` — embedded NATS JetStream. Holds in-flight events between an ingest POST and the ingest worker → ClickHouse flush, plus the `stream.gap_window_minutes` window (settings directory) of history that powers SSE gap-fill across restarts. -- `/pebble` — the Pebble dedup KV: one instance shared by every tenant, each key led by its tenant and table. Only used while some tenant's `dedupe.enabled` is `true` in its `config.json` (opened and closed on reload). +- `/pebble` — the Pebble dedup KV: one instance shared by every tenant, each key led by its tenant and table. Only used while some tenant's `dedupe.enabled` is `true` in its `config.json` (opened and closed on reload). It grows with every id kept: with `dedupe.retention` at `"0"` (forever) nothing is ever removed, so size the volume for it or set a [retention](/settings-directory#deduplication), whose expired ids an hourly sweep deletes. In a Docker / Podman / Kubernetes deployment, **`data_dir` must resolve to a host-backed volume**. The reference compose file `deployments/compose/standalone.yaml` sets `WH_DATA_DIR=/app/data` and binds a `wavehouse-data:/app/data` volume — copy that pattern. The bundled Dockerfiles pre-create `/app/data` and `/app/settings` owned by the nonroot user (UID 65532); the binary creates the `nats/` and `pebble/` subdirectories under `/app/data` itself on first run. @@ -419,7 +419,9 @@ WaveHouse discovers this schema on startup and refreshes it every `schema.refres ## Upgrading across the dedupe key change -The dedupe key now carries the table as well as the tenant ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)), so **an id deduped before the upgrade is not recognized after it**: a record carrying it is accepted once more. Nothing is migrated, and the old keys stay in `/pebble`, unread; nothing removes them yet ([#220](https://github.com/Wave-RF/WaveHouse/issues/220) tracks the sweep that will). Only a tenant with `dedupe.enabled` on is affected, and only by a record sent both before and after the upgrade — typically a producer retrying across the restart. To avoid duplicate rows, let retrying producers finish, or pause them, before upgrading. +The dedupe key now carries the table as well as the tenant ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)), so **an id deduped before the upgrade is not recognized after it**: a record carrying it is accepted once more. Nothing is migrated. The old keys are never read, and the dedupe sweep deletes them: its first pass runs about a minute after the instance opens, and `wavehouse_dedupe_swept_keys_total{reason="version_0"}` counts them ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)). Pebble returns their disk space as it compacts, not at once. Only a tenant with `dedupe.enabled` on is affected, and only by a record sent both before and after the upgrade — typically a producer retrying across the restart. To avoid duplicate rows, let retrying producers finish, or pause them, before upgrading. + +The same release adds **`dedupe.retention`, a required key**: every `config.json`, each tenant's folder included, must state it or the directory is refused (at boot) or not adopted (on reload). `"retention": "0"` keeps every id forever, as before; see [Deduplication](/settings-directory#deduplication) for a finite one. ## Upgrading across the v2 ingest envelope diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index 5a736b679..a0a62952c 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -62,7 +62,9 @@ The tell for a commit-cadence problem (ZFS-without-SLOG, noisy-neighbor VM host) With [deduplication](/settings-directory#deduplication) on, a `200` also means the records' ids were committed to the dedupe store, or, if that commit failed, that the failure was counted by `wavehouse_dedupe_commit_failed_total` and the ids lapse with their lease. On the embedded Pebble store that commit is an `fsync` of its own. It is taken once per window of up to 256 records of a request, after the window's publishes, rather than once per record: a 1,000-record batch costs four dedupe syncs, not a thousand. Measured with `BenchmarkIngest_DedupBatchOnPebble` on an Apple M4 Pro under a load average near 30, with the queue stubbed out so only the dedupe store touched disk, the dedupe work for that batch took 24 ms windowed against 5.7 s one record at a time; the JetStream publishes' own fsyncs come on top. A single-record request still pays one sync for its publish and one for its commit. -A publish can also fail after JetStream stored the event (a timeout on the ack). The record's id is then left to lapse with its 30-second dedupe lease rather than given back, and every deduped record is published under an idempotency key derived from its tenant, table and id, which each tenant's ingest stream remembers for two minutes after the first publish. A retry after the lease but inside those two minutes is therefore dropped by the stream rather than stored twice; one later than that is stored again. The 30-second lease sits well inside those two minutes, so a prompt retry is covered. +With a finite `dedupe.retention`, expired ids are deleted by a background sweep, an hour apart. Its deletes are not fsynced (a delete lost to a crash is redone by the next pass), so it adds no sync to the ingest path; it reads and deletes 1,024 keys at a time, and a commit that arrives mid-chunk waits for that chunk. An expired id is already treated as new by the next claim of it, sweep or no sweep, so retention never depends on the sweep having run. + +A publish can also fail after JetStream stored the event (a timeout on the ack). The record's id is then left to lapse with its 30-second dedupe lease rather than given back, and every deduped record is published under an idempotency key derived from its tenant, table and id, which each tenant's ingest stream remembers for two minutes after the first publish. A retry after the lease but inside those two minutes is therefore dropped by the stream rather than stored twice; one later than that is stored again. The 30-second lease sits well inside those two minutes, so a prompt retry is covered. For the same reason a finite `dedupe.retention` must be at least those two minutes: an id re-sent after a shorter retention ended would be claimed again, then dropped by the stream as a copy while the client was told it was accepted. Settings validation refuses one below it. ## Check your storage before you trust it diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 1e898fdca..126f35dd1 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -123,7 +123,8 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `dedupe.enabled` | `false` | Turn deduplication on; a reload opens or closes this tenant's store — see [Deduplication](#deduplication). | | `dedupe.id_field` | `event_id` | Dedup key field — see [Deduplication](#deduplication). | | `dedupe.require_id` | `false` | Reject rows missing the id field — see [Deduplication](#deduplication). | -| `dedupe.tables.
.{id_field, require_id}` | `{}` | Optional per-table overrides; each entry overrides only the fields it names and inherits the rest. | +| `dedupe.retention` | `"0"` | How long a committed id stays a duplicate, as a duration (`"720h"`); `"0"` keeps it forever — see [Deduplication](#deduplication). | +| `dedupe.tables.
.{id_field, require_id, retention}` | `{}` | Optional per-table overrides; each entry overrides only the fields it names and inherits the rest. | | `dlq.enabled` | `true` | Park poison rows — those that still fail after row-by-row isolation, and every row of a batch whose tenant has no ClickHouse connection — on the tenant's dead-letter stream (`DLQ_{tenant}`) (`false`: leave them unacked for redelivery — except an envelope the worker cannot read, which is dropped and counted) — see [Dead Letter Queue](#dead-letter-queue). | | `dlq.tables.
.enabled` | `{}` | Optional per-table override of the switch. | | `query.timestamp_bucket_seconds` | `60` | Bucket (seconds, `>= 0`) that a structured query's relative time range is truncated to, so near-identical queries share a cache entry; `0` disables bucketing. Read per query. | @@ -161,8 +162,9 @@ The tenant tunables. Every key is required (a missing one is a validation error) "enabled": false, "id_field": "event_id", "require_id": false, + "retention": "720h", "tables": { - "clicks": { "id_field": "click_id" } + "clicks": { "id_field": "click_id", "retention": "24h" } } }, "dlq": { @@ -188,7 +190,8 @@ Every dedupe knob lives here — there are no boot-config keys for it. The switc - `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens or closes this tenant's store in the embedded Pebble instance at `/pebble`, so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`503 dedupe store unavailable`, `Retry-After: 5`) until the next reload or restart — the files asked for dedupe, so publishing un-deduped is not a fallback. At boot a failed open refuses to start, like every other store. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) every tenant's seen ids live in that one instance, each key led by its tenant and table, and it is open while any tenant's switch is on: each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and keeps its seen ids for the folder that restores it; and if that instance fails to open, at boot or on reload, every tenant with dedupe on fails closed — its ingest answers `503 dedupe store unavailable` (`Retry-After: 5`) until a reload opens it — while the tenants with dedupe off carry on. - `dedupe.id_field` (seed default `event_id`) — JSON field name in the ingest body used as the dedup key. An id is a duplicate only within its own tenant and table: the same value in two tables is two ids. An id longer than 1,024 bytes is stored as its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. An id is committed only after its record is published; if that commit fails (counted by `wavehouse_dedupe_commit_failed_total`, which should stay at zero), the record is still answered `ok` and the id lapses with its 30-second lease: a retry of it before then answers in-flight, one inside the ingest queue's two-minute duplicate window is dropped there by its idempotency key, and one after that is stored again. - `dedupe.require_id` (seed default `false`) — controls what happens to a row missing `id_field`, or carrying it as `null` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. -- `dedupe.tables.
.{id_field, require_id}` — per-table overrides; each entry overrides only the fields it names and inherits the rest. +- `dedupe.retention` (seed default `"0"`) — how long a committed id stays a duplicate, as a Go duration string: `"24h"`, `"720h"` (30 days), `"90m"`. There is no day unit. `"0"` keeps every id forever, which was the only behavior before this key existed. Once an id's retention has ended, the next record carrying it is published as new, and a background sweep deletes the expired id from the store: first a minute after the store opens, then hourly, counted by `wavehouse_dedupe_swept_keys_total{reason="expired"}`. A finite retention must be at least `"2m"`, the ingest queue's duplicate window: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and then dropped by the queue as a copy, while the client was told it was accepted. A retention below that is refused, not raised to the minimum; so are a negative value and anything that is not a duration, such as `"30d"` or a bare number. Hot-reloadable: a change applies to ids committed after the reload, and an id already committed keeps the expiry it was stored with. +- `dedupe.tables.
.{id_field, require_id, retention}` — per-table overrides; each entry overrides only the fields it names and inherits the rest. A table can keep ids for a shorter time than its tenant, or for longer, or forever (`"retention": "0"`) under a finite tenant retention. ## ClickHouse diff --git a/internal/api/ingest.go b/internal/api/ingest.go index 6f129e34c..1189547a4 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -50,12 +50,11 @@ type IngestHandler struct { // store, picked off the store the handler already holds (#583 story 7; // dedupe.Stores in production). nil when no dedupe store is wired (tests). Dedup func(store *settings.Store) dedupe.Deduplicator - // DedupeSettings resolves the effective dedupe id_field/require_id for a - // table of the request's tenant ((*settings.Store).DedupeFor in - // production). Called once per record so a settings reload lands at a - // record boundary — one record never mixes two documents' values. Dedup is - // skipped when nil. - DedupeSettings func(store *settings.Store, table string) (enabled bool, idField string, requireID bool) + // DedupeSettings resolves the effective dedupe settings for a table of the + // request's tenant ((*settings.Store).DedupeFor in production). Called + // once per record so a settings reload lands at a record boundary — one + // record never mixes two documents' values. Dedup is skipped when nil. + DedupeSettings func(store *settings.Store, table string) settings.Dedupe // DedupeLease is how long a record's claimed id stays pending while it is // published; 0 means dedupe.DefaultLease. DedupeLease time.Duration @@ -572,8 +571,10 @@ type pendingRecord struct { reject *recordReject // non-nil: the record is bad and is not published payload []byte // the encoded envelope to publish // key is the record's dedupe identity, nil when it is published - // un-deduped; claim is Reserve's answer for it. + // un-deduped; retention is how long its id stays a duplicate once + // committed; claim is Reserve's answer for it. key *dedupe.Key + retention time.Duration claim dedupe.Claim duplicate bool } @@ -701,7 +702,7 @@ func (h *IngestHandler) prepareRecord( // enforces) after the permission checks: check clauses keep pre-#372 semantics. h.validator().CanonicalizeTimestamps(schema, data) - // Optional deduplication. enabled/id_field/require_id resolve per record + // Optional deduplication. The dedupe settings resolve per record // from one snapshot (table override → global; the settings directory // always states them, so no compiled fallback is needed), so a reload // lands at a record boundary. A Deduplicator without a settings source is @@ -709,14 +710,16 @@ func (h *IngestHandler) prepareRecord( // in ingestWindow, once every record of the window is encoded, so nothing // but the publish can fail while the claim is held. if h.Dedup != nil && h.DedupeSettings != nil { - if enabled, idField, requireID := h.DedupeSettings(store, table); enabled { + if dd := h.DedupeSettings(store, table); dd.Enabled { + idField := dd.IDField // An explicit null is as missing as an absent key (#370): fmt.Sprint // would make every null "", one id for every such record. if idVal, ok := data[idField]; ok && idVal != nil { rec.key = &dedupe.Key{Table: table, ID: fmt.Sprint(idVal)} + rec.retention = dd.Retention } else { dedupeMissingIDCounter.Add(ctx, 1, metric.WithAttributes(attribute.String("table", table))) - if requireID { + if dd.RequireID { slog.WarnContext(ctx, "dedupe id_field missing or null; rejecting", "id_field", idField, "table", table) return pendingRecord{reject: &recordReject{ Status: http.StatusBadRequest, @@ -793,7 +796,7 @@ func (h *IngestHandler) ingestWindow(ctx context.Context, store *settings.Store, return h.publishFailed(ctx, dd, topic, recs, i, err) } } - commitClaims(ctx, dd, claimedIn(recs), table) + commitClaims(ctx, dd, recs, table) return nil } @@ -865,7 +868,7 @@ func (h *IngestHandler) reserve(ctx context.Context, dd dedupe.Deduplicator, tab // the first copy landed. The records after k were never sent and are released. func (h *IngestHandler) publishFailed(ctx context.Context, dd dedupe.Deduplicator, topic mq.Topic, recs []pendingRecord, k int, err error) *requestAbort { definite := errors.Is(err, mq.ErrQueueFull) - commitClaims(ctx, dd, claimedIn(recs[:k]), topic.Table) + commitClaims(ctx, dd, recs[:k], topic.Table) after := k + 1 if definite { after = k @@ -890,20 +893,33 @@ func claimedIn(recs []pendingRecord) []dedupe.Claim { return out } -// commitClaims makes published records' ids duplicates. A failure does not -// fail the records — they are in the queue — so it is logged and counted, and -// the claims lapse after their lease. -func commitClaims(ctx context.Context, dd dedupe.Deduplicator, claims []dedupe.Claim, table string) { - if len(claims) == 0 { - return +// commitClaims makes the ids of recs' Claimed claims duplicates, one Commit +// per retention — one in practice, unless a reload changed it mid-window. A +// failure does not fail the records — they are in the queue — so it is logged +// and counted, and the claims lapse after their lease. +func commitClaims(ctx context.Context, dd dedupe.Deduplicator, recs []pendingRecord, table string) { + var retentions []time.Duration + byRetention := map[time.Duration][]dedupe.Claim{} + for i := range recs { + if recs[i].claim.Status != dedupe.Claimed { + continue + } + r := recs[i].retention + if _, ok := byRetention[r]; !ok { + retentions = append(retentions, r) + } + byRetention[r] = append(byRetention[r], recs[i].claim) } - // The records are queued whatever the request's context does next. - err := dd.Commit(context.WithoutCancel(ctx), claims, 0) - switch { - case err == nil, errors.Is(err, dedupe.ErrDisabled): - default: - dedupeCommitFailedCounter.Add(ctx, int64(len(claims)), metric.WithAttributes(attribute.String("table", table))) - slog.ErrorContext(ctx, "dedupe commit failed after publish; the ids lapse with their lease", "error", err, "table", table, "records", len(claims)) + for _, r := range retentions { + claims := byRetention[r] + // The records are queued whatever the request's context does next. + err := dd.Commit(context.WithoutCancel(ctx), claims, r) + switch { + case err == nil, errors.Is(err, dedupe.ErrDisabled): + default: + dedupeCommitFailedCounter.Add(ctx, int64(len(claims)), metric.WithAttributes(attribute.String("table", table))) + slog.ErrorContext(ctx, "dedupe commit failed after publish; the ids lapse with their lease", "error", err, "table", table, "records", len(claims)) + } } } diff --git a/internal/api/ingest_retention_test.go b/internal/api/ingest_retention_test.go new file mode 100644 index 000000000..ef65e4b2f --- /dev/null +++ b/internal/api/ingest_retention_test.go @@ -0,0 +1,104 @@ +package api + +import ( + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/mq" + "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil" +) + +// A finite retention must outlast the queue's duplicate window, or an id +// re-sent after it expires is claimed again and then dropped by the queue as +// a copy of the first publish. +func TestIngest_MinDedupeRetentionCoversTheDuplicateWindow(t *testing.T) { + t.Parallel() + assert.GreaterOrEqual(t, settings.MinDedupeRetention, mq.EmbeddedDuplicateWindow) +} + +// dedupeConfig is fullConfig with dedupe switched on and the given dedupe +// block's retention settings. +func dedupeConfig(retention, tables string) string { + return strings.Replace(fullConfig(100), + `"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}`, + `"dedupe": {"enabled": true, "id_field": "event_id", "require_id": false, "retention": "`+retention+`", "tables": `+tables+`}`, 1) +} + +// Each record is committed with its table's retention from the adopted +// settings, and a reload changes it for the next request: the retention is +// read per record, like id_field, not fixed when the store was opened. +func TestIngest_Dedup_CommitsWithTheAdoptedRetention(t *testing.T) { + t.Parallel() + dir := writeSettingsFixture(t, dedupeConfig("720h", `{"users": {"retention": "0"}}`)) + tenants, findings := settings.Open(dir) + require.NotNil(t, tenants, "findings: %v", findings) + store, _ := tenants.For(tenant.Default) + + reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ + {Name: "clicks", Columns: []discovery.Column{{Name: "event_id", Type: "String"}}}, + {Name: "users", Columns: []discovery.Column{{Name: "event_id", Type: "String"}}}, + }) + dedup := testutil.NewMockDeduplicator() + h := NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}) + h.Dedup = staticDedup(dedup) + h.DedupeSettings = (*settings.Store).DedupeFor + ingest := func(table, id string) { + t.Helper() + w := httptest.NewRecorder() + req := ingestRequest(t, table, map[string]any{"event_id": id}) + h.Handle(w, req.WithContext(WithStore(req.Context(), store))) + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + } + + ingest("clicks", "e1") + ingest("users", "e1") + assert.Equal(t, 720*time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "e1"})) + assert.Equal(t, time.Duration(0), dedup.Retention(dedupe.Key{Table: "users", ID: "e1"}), "the table keeps ids forever") + + require.NoError(t, os.WriteFile(filepath.Join(dir, settings.FileConfig), []byte(dedupeConfig("24h", `{}`)), 0o600)) + _, adopted := tenants.Reload("test") + require.True(t, adopted) + ingest("clicks", "e2") + ingest("users", "e2") + assert.Equal(t, 24*time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "e2"})) + assert.Equal(t, 24*time.Hour, dedup.Retention(dedupe.Key{Table: "users", ID: "e2"}), "the override is gone") + assert.Equal(t, 720*time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "e1"}), "ids committed before the change keep theirs") +} + +// A reload that lands mid-window splits the window's commit by retention, so +// every record keeps the retention of the snapshot it was prepared under. +func TestIngest_Dedup_ReloadMidWindowCommitsEachRetention(t *testing.T) { + t.Parallel() + dedup := testutil.NewMockDeduplicator() + h := NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}) + h.Dedup = staticDedup(dedup) + var calls atomic.Int32 + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + if calls.Add(1) <= 2 { + return settings.Dedupe{Enabled: true, IDField: "event_id", Retention: time.Hour} + } + return settings.Dedupe{Enabled: true, IDField: "event_id", Retention: 2 * time.Hour} + } + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", + `{"page": "/", "event_id": "a"}`, `{"page": "/", "event_id": "b"}`, `{"page": "/", "event_id": "c"}`))) + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + assert.Equal(t, 2, dedup.Commits, "one Commit per retention") + assert.Equal(t, time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "a"})) + assert.Equal(t, time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "b"})) + assert.Equal(t, 2*time.Hour, dedup.Retention(dedupe.Key{Table: "clicks", ID: "c"})) +} diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index 579933339..763ba2838 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -201,7 +201,9 @@ func TestIngest_Dedup_FirstTime(t *testing.T) { dedup := testutil.NewMockDeduplicator() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "evt-1"}) w := httptest.NewRecorder() @@ -217,7 +219,9 @@ func TestIngest_Dedup_Duplicate(t *testing.T) { dedup := testutil.NewMockDeduplicator() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } // First call. req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "event_id": "dup-1"}) @@ -703,7 +707,9 @@ func TestIngest_DedupIsTheTenants(t *testing.T) { pub := &testutil.MockPublisher{} h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = func(s *settings.Store) dedupe.Deduplicator { return stores.For(s.Tenant()) } - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } ingest := func(id tenant.ID) string { store, ok := tenants.For(id) @@ -726,7 +732,9 @@ func TestIngest_Dedup_MissingIDField(t *testing.T) { dedup := testutil.NewMockDeduplicator() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } // Payload omits event_id and require_id is off: the row skips // dedup and is still published — the warn+counter path, not a rejection (#219). @@ -745,7 +753,9 @@ func TestIngest_Dedup_RequireID_Rejects(t *testing.T) { pub := &testutil.MockPublisher{} h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(testutil.NewMockDeduplicator()) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} + } w := httptest.NewRecorder() h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home"}))) @@ -767,7 +777,9 @@ func TestIngest_NDJSON_RequireID_Rejects(t *testing.T) { pub := &testutil.MockPublisher{} h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(testutil.NewMockDeduplicator()) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} + } req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a", "event_id": "e1"}), @@ -1015,7 +1027,9 @@ func TestIngest_NDJSON_Dedup(t *testing.T) { dedup := testutil.NewMockDeduplicator() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a", "event_id": "e1"}), @@ -2306,7 +2320,9 @@ func TestIngest_Dedup_DisabledBySettings(t *testing.T) { dedup.Err = errors.New("must not be called while disabled") h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return false, "event_id", true } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{IDField: "event_id", RequireID: true} + } w := httptest.NewRecorder() h.Handle(w, withTenant(ingestRequest(t, "clicks", tt.body))) @@ -2326,7 +2342,9 @@ func TestIngest_Dedup_DisabledMidReload(t *testing.T) { dedup.Err = dedupe.ErrDisabled h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} + } w := httptest.NewRecorder() h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"event_id": "e1", "page": "/home"}))) @@ -2744,7 +2762,9 @@ func dedupHandler(t *testing.T, pub *testutil.MockPublisher, dedup dedupe.Dedupl t.Helper() h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", requireID } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: requireID} + } return h } diff --git a/internal/api/ingest_window_test.go b/internal/api/ingest_window_test.go index 3d1dc931a..32015dc48 100644 --- a/internal/api/ingest_window_test.go +++ b/internal/api/ingest_window_test.go @@ -391,7 +391,9 @@ func pebbleBatchHandler(tb testing.TB, window int) (*IngestHandler, *countingDed counted := &countingDedup{Deduplicator: store} h := NewIngestHandler(fixedRegistry(testRegistry(tb)), &testutil.MockPublisher{}) h.Dedup = staticDedup(counted) - h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } h.window = window return h, counted } diff --git a/internal/api/settings_test.go b/internal/api/settings_test.go index 9e823ceaa..24fa80f12 100644 --- a/internal/api/settings_test.go +++ b/internal/api/settings_test.go @@ -19,7 +19,7 @@ import ( // fullConfig is a complete config.json (every key is required) with the // given query.default_max_rows. func fullConfig(maxRows int) string { - return fmt.Sprintf(`{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": %d, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, maxRows) + return fmt.Sprintf(`{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": %d, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, maxRows) } // writeSettingsFixture materializes a minimal valid settings directory whose diff --git a/internal/app/app_test.go b/internal/app/app_test.go index 1468d0df5..019214c0a 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -208,7 +208,7 @@ func TestNew_DedupeFollowsSettings(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { dir := writeSettings(t, map[string]any{"dedupe": map[string]any{ - "enabled": tt.enabled, "id_field": "event_id", "require_id": false, "tables": map[string]any{}, + "enabled": tt.enabled, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}, }}) cfg := testConfig(t, dir) a := newApp(t, cfg, Options{}) @@ -241,7 +241,7 @@ func TestReload_DrivesTheRegisteredHooks(t *testing.T) { require.Equal(t, int64(1<<30), a.mq.MaxBytes(tenant.Default)) rewriteSettings(t, dir, map[string]any{ - "dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}, + "dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}}, "mq": map[string]any{"max_bytes_gb": 2}, }) _, adopted := a.tenants.Reload("test") @@ -405,7 +405,7 @@ func TestNew_NestedWithoutAnOperatorKeyWarnsTheOpsTreeIsClosed(t *testing.T) { // request, so a lost 0 folder is felt at once on the routes that read tenant // 0's list. func TestReload_NestedHooksFollowEachTenant(t *testing.T) { - dedupeOn := map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}} + dedupeOn := map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}} grown := map[string]any{"dedupe": dedupeOn, "mq": map[string]any{"max_bytes_gb": 2}} root := writeNestedSettings(t, map[string]map[string]any{ "0": {"mq": map[string]any{"max_bytes_gb": 1}}, @@ -478,7 +478,7 @@ func TestReload_NestedHooksFollowEachTenant(t *testing.T) { // reopened over the same seen ids when the folder is back. The instance is // open while some tenant's store is, and Close releases it. func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { - dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}} + dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}}} root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn, "globex": nil, "broken": invalidQuery}) cfg := testConfig(t, root) a := newApp(t, cfg, Options{}) @@ -542,7 +542,7 @@ func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { // instance — their ingest answers 500 until a reload or a restart opens it — // while the process, and every tenant with dedupe off, carries on. func TestNew_DedupeOpenFailure(t *testing.T) { - dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}} + dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}}} // A regular file where the instance's directory should be is what Pebble // refuses to open. block := func(t *testing.T, dataDir string) { @@ -852,7 +852,7 @@ func analystPipe(t *testing.T, dir string) { func TestNew_LateBootFailureReleasesEverything(t *testing.T) { guardGlobals(t) dir := writeSettings(t, map[string]any{"dedupe": map[string]any{ - "enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}, + "enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}, }}) cfg := testConfig(t, dir) natsDir := filepath.Join(cfg.DataDir, "nats") @@ -1458,7 +1458,7 @@ func TestReload_CeilingRefusesAThirdTupleThenOpensIt(t *testing.T) { func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { jwks, _, fetches := jwksServer(t, "acme-1") acmeSettings := authPatch(jwks.URL) - acmeSettings["dedupe"] = map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}} + acmeSettings["dedupe"] = map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}} root := writeNestedSettings(t, map[string]map[string]any{"acme": acmeSettings, "globex": nil}) a := newApp(t, testConfig(t, root), Options{}) acme, acmeRegistry, acmeDedup := a.pools.For("acme"), a.discoveries.For("acme"), a.dedup.For("acme") diff --git a/internal/dedupe/embedded.go b/internal/dedupe/embedded.go index b3f015f0d..5186a7380 100644 --- a/internal/dedupe/embedded.go +++ b/internal/dedupe/embedded.go @@ -30,9 +30,16 @@ import ( type Embedded struct { dir string - mu sync.Mutex // guards db and open - db *pebble.DB - open int // tenant stores open over db + mu sync.Mutex // guards db, open and stopSweep + db *pebble.DB + open int // tenant stores open over db + stopSweep func() // stops db's sweep + + // commitMu is read-held by Commit and held by a sweep chunk, so a sweep + // never deletes a key a Commit rewrote after the sweep read it. + commitMu sync.RWMutex + sweepFirst time.Duration + sweepEvery time.Duration pending *pendingSet tokens atomic.Uint64 @@ -45,7 +52,13 @@ type Embedded struct { // NewEmbedded returns the embedded implementation under dataDir. Nothing is // opened until a tenant's store is. func NewEmbedded(dataDir string) *Embedded { - return &Embedded{dir: filepath.Join(dataDir, "pebble"), pending: newPendingSet(), now: time.Now} + return &Embedded{ + dir: filepath.Join(dataDir, "pebble"), + pending: newPendingSet(), + now: time.Now, + sweepFirst: sweepFirstDelay, + sweepEvery: sweepInterval, + } } // Dir is where the instance lives. @@ -76,6 +89,7 @@ func (e *Embedded) acquire(prefix []byte) (Deduplicator, error) { return nil, err } e.db = db + e.stopSweep = e.startSweep(db) } e.open++ return &tenantStore{e: e, db: e.db, prefix: prefix}, nil @@ -90,6 +104,7 @@ func (e *Embedded) release() error { if e.open > 0 { return nil } + e.stopSweep() err := e.db.Close() e.db = nil return err @@ -182,24 +197,36 @@ func (s *tenantStore) reserve(key []byte, k Key, now time.Time, lease time.Durat // committedLive reports whether a stored value is a commit that has not // expired. func committedLive(val []byte, now time.Time) bool { + exp, ok := committedExpiry(val) + return ok && (exp == 0 || now.UnixNano() < exp) +} + +// committedExpired reports whether a stored value is a commit whose +// retention has ended — what the sweep deletes. +func committedExpired(val []byte, now time.Time) bool { + exp, ok := committedExpiry(val) + return ok && exp != 0 && now.UnixNano() >= exp +} + +// committedExpiry reads a commit's expiry (UnixNano, 0 = never); ok is false +// for a value that is not a commit. +func committedExpiry(val []byte) (exp int64, ok bool) { if len(val) != valueLen || val[0] != committedMark { - return false + return 0, false } - exp := int64(binary.BigEndian.Uint64(val[1:])) //nolint:gosec // written from an int64 below - return exp == 0 || now.UnixNano() < exp + return int64(binary.BigEndian.Uint64(val[1:])), true //nolint:gosec // written from an int64 below } // Commit writes every claim in one batch and one fsync, then drops the // pending entries it still owns — in that order, so no Reserve in between // finds the key neither pending nor committed. func (s *tenantStore) Commit(_ context.Context, claims []Claim, retention time.Duration) error { - var exp int64 - if retention > 0 { - exp = s.e.now().Add(retention).UnixNano() - } + s.e.commitMu.RLock() + defer s.e.commitMu.RUnlock() + exp := expiry(s.e.now(), retention) val := make([]byte, valueLen) val[0] = committedMark - binary.BigEndian.PutUint64(val[1:], uint64(exp)) + binary.BigEndian.PutUint64(val[1:], uint64(exp)) //nolint:gosec // expiry is never negative b := s.db.NewBatch() defer func() { _ = b.Close() }() for _, c := range claims { @@ -214,6 +241,20 @@ func (s *tenantStore) Commit(_ context.Context, claims []Claim, retention time.D return nil } +// expiry is the stored expiry of a commit at now kept for retention: 0 for +// none, and the latest representable instant for a retention reaching past +// it, rather than a wrapped-around one in the past. +func expiry(now time.Time, retention time.Duration) int64 { + if retention <= 0 { + return 0 + } + n := now.UnixNano() + if retention > time.Duration(math.MaxInt64-n) { + return math.MaxInt64 + } + return n + int64(retention) +} + // Release drops the pending entries the claims still own. func (s *tenantStore) Release(_ context.Context, claims []Claim) error { s.release(claims) diff --git a/internal/dedupe/sweep.go b/internal/dedupe/sweep.go new file mode 100644 index 000000000..e536c5edc --- /dev/null +++ b/internal/dedupe/sweep.go @@ -0,0 +1,151 @@ +package dedupe + +import ( + "bytes" + "context" + "fmt" + "log/slog" + "time" + + "github.com/cockroachdb/pebble" + "go.opentelemetry.io/otel" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/metric" +) + +// The sweep's cadence. Expired keys are already absent to Reserve, so the +// sweep only reclaims space and can run rarely; the first pass comes soon +// after the instance opens so an upgrade's version-0 keys go without waiting +// an hour. +const ( + sweepInterval = time.Hour + sweepFirstDelay = time.Minute + // sweepChunk keys are read and deleted per lock hold, with sweepPause + // between chunks: at most ~100k keys a second, and a Commit never waits + // longer than one chunk. + sweepChunk = 1024 + sweepPause = 10 * time.Millisecond +) + +// Swept-key reasons, the metric's reason attribute. +const ( + sweptExpired = "expired" + sweptVersion0 = "version_0" + sweptAttribute = "reason" +) + +var sweptKeysCounter, _ = otel.Meter("wavehouse-dedupe").Int64Counter( + "wavehouse_dedupe_swept_keys_total", + metric.WithDescription("Keys the embedded dedupe sweep deleted, by reason: expired (retention ended) or version_0 (the layout before ids were keyed by table)"), +) + +// sweepResult is what a sweep deleted. +type sweepResult struct { + Expired, Version0 int +} + +// startSweep runs the sweep over db until the returned stop is called; stop +// waits for a chunk in progress to finish. Callers hold e.mu. +func (e *Embedded) startSweep(db *pebble.DB) (stop func()) { + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { + defer close(done) + wait := e.sweepFirst + for { + select { + case <-ctx.Done(): + return + case <-time.After(wait): + } + wait = e.sweepEvery + res, err := e.sweep(ctx, db) + switch { + case err != nil && ctx.Err() == nil: + slog.WarnContext(ctx, "dedupe sweep failed; retrying next interval", "error", err, "expired", res.Expired, "version_0", res.Version0) + case res.Expired+res.Version0 > 0: + slog.InfoContext(ctx, "dedupe sweep deleted keys", "expired", res.Expired, "version_0", res.Version0) + } + } + }() + return func() { + cancel() + <-done + } +} + +// sweep makes one pass over the whole instance, deleting keys whose +// retention has ended and version-0 keys (tenant ‖ 0x00 ‖ id, from before +// ids were keyed by table), which nothing reads. It stops early, without +// error, when ctx ends. +func (e *Embedded) sweep(ctx context.Context, db *pebble.DB) (sweepResult, error) { + var res sweepResult + var from []byte + for { + next, err := e.sweepChunk(ctx, db, from, &res) + if err != nil || next == nil { + return res, err + } + from = next + select { + case <-ctx.Done(): + return res, nil + case <-time.After(sweepPause): + } + } +} + +// sweepChunk deletes the sweepable keys among the next sweepChunk keys from +// from, returning where the next chunk starts (nil at the end). It holds +// commitMu, so no Commit lands between reading a key and deleting it: a key +// re-committed after it expired is never deleted with its new value. +func (e *Embedded) sweepChunk(ctx context.Context, db *pebble.DB, from []byte, res *sweepResult) ([]byte, error) { + e.commitMu.Lock() + defer e.commitMu.Unlock() + now := e.now() + it, err := db.NewIter(&pebble.IterOptions{LowerBound: from}) + if err != nil { + return nil, fmt.Errorf("dedupe sweep: %w", err) + } + b := db.NewBatch() + defer func() { _ = b.Close() }() + var next []byte + var expired, v0 int64 + seen := 0 + for valid := it.First(); valid; valid = it.Next() { + if seen == sweepChunk { + next = bytes.Clone(it.Key()) + break + } + seen++ + k := it.Key() + switch { + case len(k) == 0 || k[0] != keyVersion: + v0++ + case committedExpired(it.Value(), now): + expired++ + default: + continue + } + if err := b.Delete(k, nil); err != nil { + _ = it.Close() + return nil, fmt.Errorf("dedupe sweep: %w", err) + } + } + if err := it.Close(); err != nil { + return nil, fmt.Errorf("dedupe sweep: %w", err) + } + // NoSync: a delete lost to a crash is redone by the next pass. + if err := b.Commit(pebble.NoSync); err != nil { + return nil, fmt.Errorf("dedupe sweep: %w", err) + } + res.Expired += int(expired) + res.Version0 += int(v0) + if expired > 0 { + sweptKeysCounter.Add(ctx, expired, metric.WithAttributes(attribute.String(sweptAttribute, sweptExpired))) + } + if v0 > 0 { + sweptKeysCounter.Add(ctx, v0, metric.WithAttributes(attribute.String(sweptAttribute, sweptVersion0))) + } + return next, nil +} diff --git a/internal/dedupe/sweep_test.go b/internal/dedupe/sweep_test.go new file mode 100644 index 000000000..44579dcd8 --- /dev/null +++ b/internal/dedupe/sweep_test.go @@ -0,0 +1,158 @@ +package dedupe + +import ( + "context" + "errors" + "fmt" + "math" + "sync/atomic" + "testing" + "time" + + "github.com/cockroachdb/pebble" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// stepClock is a clock a test moves by hand. +type stepClock struct{ ns atomic.Int64 } + +func newStepClock() *stepClock { + c := &stepClock{} + c.ns.Store(time.Now().UnixNano()) + return c +} + +func (c *stepClock) now() time.Time { return time.Unix(0, c.ns.Load()) } +func (c *stepClock) advance(d time.Duration) { c.ns.Add(int64(d)) } +func present(t *testing.T, e *Embedded, key []byte) bool { + t.Helper() + _, closer, err := e.db.Get(key) + if errors.Is(err, pebble.ErrNotFound) { + return false + } + require.NoError(t, err) + _ = closer.Close() + return true +} + +// commitIDs reserves and commits ids in table "events" with retention. +func commitIDs(t *testing.T, m *Managed, retention time.Duration, ids ...string) { + t.Helper() + keys := make([]Key, len(ids)) + for i, id := range ids { + keys[i] = Key{Table: "events", ID: id} + } + claims, err := m.Reserve(context.Background(), keys, DefaultLease) + require.NoError(t, err) + require.NoError(t, m.Commit(context.Background(), claims, retention)) +} + +// A sweep deletes the keys whose retention has ended and every version-0 +// key, across chunk boundaries, and leaves every live key: one kept forever, +// one not yet expired, and one that expired and was committed again. +func TestEmbedded_SweepDeletesExpiredAndVersionZeroKeys(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + clock := newStepClock() + SetClock(e, clock.now) + acme, globex := switchedOn(t, e, "acme"), switchedOn(t, e, "globex") + + // More expired keys than one chunk holds, interleaved with live ones. + var expired, live []string + for i := range 2*sweepChunk + 10 { + expired = append(expired, fmt.Sprintf("x%05d", i)) + live = append(live, fmt.Sprintf("x%05d-live", i)) + } + commitIDs(t, acme, time.Hour, expired...) + commitIDs(t, acme, 3*time.Hour, live...) + commitIDs(t, globex, 0, "forever") + commitIDs(t, globex, time.Hour, "recommitted") + for _, k := range []string{"acme\x00e1", "acme\x00e2", "globex\x00e1"} { + require.NoError(t, e.db.Set([]byte(k), make([]byte, 8), pebble.Sync)) + } + + clock.advance(2 * time.Hour) + commitIDs(t, globex, time.Hour, "recommitted") + res, err := e.sweep(context.Background(), e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{Expired: len(expired), Version0: 3}, res) + + for _, id := range expired { + require.False(t, present(t, e, AppendKey(nil, KeyPrefix("acme"), Key{Table: "events", ID: id})), id) + } + for _, id := range live { + require.True(t, present(t, e, AppendKey(nil, KeyPrefix("acme"), Key{Table: "events", ID: id})), id) + } + assert.True(t, present(t, e, AppendKey(nil, KeyPrefix("globex"), Key{Table: "events", ID: "forever"}))) + assert.True(t, present(t, e, AppendKey(nil, KeyPrefix("globex"), Key{Table: "events", ID: "recommitted"}))) + assert.False(t, present(t, e, []byte("acme\x00e1"))) + assert.False(t, present(t, e, []byte("globex\x00e1"))) + + dup, err := mark(context.Background(), globex, "recommitted") + require.NoError(t, err) + assert.True(t, dup, "the new commit survived the sweep") + res, err = e.sweep(context.Background(), e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{}, res, "a second pass finds nothing") +} + +// A retention is honoured on read before any sweep has run: the key is a +// duplicate until the retention ends and claimable from that instant. +func TestEmbedded_RetentionHonouredOnRead(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + clock := newStepClock() + SetClock(e, clock.now) + m := switchedOn(t, e, "acme") + commitIDs(t, m, time.Hour, "e1") + + clock.advance(time.Hour - time.Nanosecond) + claims, err := m.Reserve(context.Background(), []Key{{Table: "events", ID: "e1"}}, DefaultLease) + require.NoError(t, err) + assert.Equal(t, Duplicate, claims[0].Status) + + clock.advance(time.Nanosecond) + claims, err = m.Reserve(context.Background(), []Key{{Table: "events", ID: "e1"}}, DefaultLease) + require.NoError(t, err) + assert.Equal(t, Claimed, claims[0].Status) +} + +// The sweep runs on its own once the instance opens, and stops with it. +func TestEmbedded_SweepRunsWhileOpen(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + e.sweepFirst, e.sweepEvery = time.Millisecond, time.Millisecond + m := e.Tenant("acme") + require.NoError(t, m.Apply(true)) + require.NoError(t, e.db.Set([]byte("acme\x00e1"), make([]byte, 8), pebble.Sync)) + assert.Eventually(t, func() bool { return !present(t, e, []byte("acme\x00e1")) }, 5*time.Second, 5*time.Millisecond) + require.NoError(t, m.Apply(false), "closing waits for the sweep to stop") + assert.False(t, e.Open()) +} + +// A sweep stops between chunks when its context ends. +func TestEmbedded_SweepStopsWhenCancelled(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + switchedOn(t, e, "acme") + b := e.db.NewBatch() + for i := range 3 * sweepChunk { + require.NoError(t, b.Set(fmt.Appendf(nil, "acme\x00%05d", i), nil, nil)) + } + require.NoError(t, b.Commit(pebble.Sync)) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + res, err := e.sweep(ctx, e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{Version0: sweepChunk}, res, "one chunk, then the cancellation is seen") +} + +func TestExpiry(t *testing.T) { + t.Parallel() + now := time.Unix(0, 1_000) + assert.Zero(t, expiry(now, 0)) + assert.Zero(t, expiry(now, -time.Second)) + assert.Equal(t, 1_000+int64(time.Hour), expiry(now, time.Hour)) + assert.Equal(t, int64(math.MaxInt64), expiry(now, time.Duration(math.MaxInt64)), "saturates rather than wrapping into the past") +} diff --git a/internal/settings/registry_test.go b/internal/settings/registry_test.go index 06e2a3b9a..3ac222ccb 100644 --- a/internal/settings/registry_test.go +++ b/internal/settings/registry_test.go @@ -113,9 +113,7 @@ func TestRegistry_SurvivesVanishedDirectory(t *testing.T) { assert.False(t, adopted) assert.True(t, HasErrors(findings)) assert.Equal(t, 42, s.DefaultMaxRows()) - _, id, req := s.DedupeFor("clicks") - assert.Equal(t, "event_id", id) - assert.False(t, req) + assert.Equal(t, "event_id", s.DedupeFor("clicks").IDField) } // TestRegistry_AfterAdoptRunsOnlyOnAdoption pins the lifecycle hook contract diff --git a/internal/settings/seed/config.json b/internal/settings/seed/config.json index a8ab41a24..61aa8fee3 100644 --- a/internal/settings/seed/config.json +++ b/internal/settings/seed/config.json @@ -26,6 +26,7 @@ "enabled": false, "id_field": "event_id", "require_id": false, + "retention": "0", "tables": {} }, "dlq": { diff --git a/internal/settings/settings.go b/internal/settings/settings.go index 7da6c3faa..68445543a 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -13,6 +13,8 @@ package settings import ( + "time" + "github.com/Wave-RF/WaveHouse/internal/pipes" "github.com/Wave-RF/WaveHouse/internal/policy" ) @@ -141,14 +143,18 @@ type AuthConfig struct { // (dedupe.Managed, one per tenant, each a share of the one embedded Pebble // instance), so the whole block is tenant-owned. // -// id_field and require_id are required here and optional per table: a table -// override inherits whichever field it doesn't name. An empty, +// id_field, require_id and retention are required here and optional per +// table: a table override inherits whichever field it doesn't name. An empty, // whitespace-only, or whitespace-padded id_field is rejected at every level, // so the effective id_field can never be empty or silently unmatchable. type DedupeConfig struct { Enabled *bool `json:"enabled"` IDField *string `json:"id_field"` RequireID *bool `json:"require_id"` + // Retention is how long a committed id stays a duplicate, as a Go + // duration ("720h"); "0" keeps it forever. A change applies to ids + // committed after it. + Retention *string `json:"retention"` // Tables holds per-table overrides keyed by ClickHouse table name (#222). // Names are format-checked only — existence is schema discovery's runtime // concern, same as policies.json table keys. @@ -160,8 +166,16 @@ type DedupeConfig struct { type TableDedupe struct { IDField *string `json:"id_field,omitempty"` RequireID *bool `json:"require_id,omitempty"` + Retention *string `json:"retention,omitempty"` } +// MinDedupeRetention is the shortest finite dedupe retention: the embedded +// queue's duplicate window (mq.EmbeddedDuplicateWindow). A record is +// published under an idempotency key derived from its id, so an id re-sent +// after a shorter retention but inside the window is claimed again and then +// dropped by the queue as a copy, while the client is told it was accepted. +const MinDedupeRetention = 2 * time.Minute + // DLQConfig gates the Dead Letter Queue: whether a row that still fails // after the row-by-row isolation retry is parked on the tenant's dead-letter // queue (and its original acked) or left unacked to be redelivered diff --git a/internal/settings/store.go b/internal/settings/store.go index f68a2fbf3..b3615efef 100644 --- a/internal/settings/store.go +++ b/internal/settings/store.go @@ -76,23 +76,38 @@ func (s *Store) DedupeEnabled() bool { return *s.doc().Config.Dedupe.Enabled } +// Dedupe is a table's effective dedupe settings. +type Dedupe struct { + Enabled bool + IDField string + RequireID bool + // Retention is how long a committed id stays a duplicate; 0 is forever. + Retention time.Duration +} + // DedupeFor resolves the effective dedupe settings for a table: the switch, // then the table override for each field it names, the global value -// otherwise. All three resolve from one snapshot load, so a reload can never -// hand a record the id_field of one document and the require_id (or enabled) -// of another. -func (s *Store) DedupeFor(table string) (enabled bool, idField string, requireID bool) { +// otherwise. Every field resolves from one snapshot load, so a reload can +// never hand a record the id_field of one document and the require_id, +// retention or switch of another. +func (s *Store) DedupeFor(table string) Dedupe { d := s.doc().Config.Dedupe - enabled, idField, requireID = *d.Enabled, *d.IDField, *d.RequireID + out := Dedupe{Enabled: *d.Enabled, IDField: *d.IDField, RequireID: *d.RequireID} + retention := *d.Retention if td, ok := d.Tables[table]; ok { if td.IDField != nil { - idField = *td.IDField + out.IDField = *td.IDField } if td.RequireID != nil { - requireID = *td.RequireID + out.RequireID = *td.RequireID + } + if td.Retention != nil { + retention = *td.Retention } } - return enabled, idField, requireID + // Validate has parsed it already. + out.Retention, _ = time.ParseDuration(retention) + return out } // ClickHouse is the adopted connection wiring, resolved as one value from diff --git a/internal/settings/store_test.go b/internal/settings/store_test.go index abcc6c902..e96898247 100644 --- a/internal/settings/store_test.go +++ b/internal/settings/store_test.go @@ -46,23 +46,22 @@ func TestStore_Tenant(t *testing.T) { func TestStore_DedupeFor_Cascade(t *testing.T) { t.Parallel() s := newLoadedStore(t, map[string]string{ - FileConfig: configJSON(`{"dedupe": {"require_id": true, "tables": {"clicks": {"id_field": "click_id"}, "views": {"require_id": false}}}}`), + FileConfig: configJSON(`{"dedupe": {"require_id": true, "retention": "720h", "tables": {"clicks": {"id_field": "click_id"}, "views": {"require_id": false, "retention": "24h"}, "audit": {"retention": "0"}}}}`), }) tests := []struct { - name, table, wantID string - wantRequire bool + name, table string + want Dedupe }{ - {name: "table overrides id_field, inherits require_id", table: "clicks", wantID: "click_id", wantRequire: true}, - {name: "table overrides require_id, inherits id_field", table: "views", wantID: "event_id", wantRequire: false}, - {name: "unlisted table gets globals", table: "other", wantID: "event_id", wantRequire: true}, + {name: "table overrides id_field, inherits the rest", table: "clicks", want: Dedupe{IDField: "click_id", RequireID: true, Retention: 720 * time.Hour}}, + {name: "table overrides require_id and retention, inherits id_field", table: "views", want: Dedupe{IDField: "event_id", Retention: 24 * time.Hour}}, + {name: "table keeps ids forever under a finite tenant retention", table: "audit", want: Dedupe{IDField: "event_id", RequireID: true}}, + {name: "unlisted table gets globals", table: "other", want: Dedupe{IDField: "event_id", RequireID: true, Retention: 720 * time.Hour}}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { t.Parallel() - _, id, req := s.DedupeFor(tt.table) - assert.Equal(t, tt.wantID, id) - assert.Equal(t, tt.wantRequire, req) + assert.Equal(t, tt.want, s.DedupeFor(tt.table)) }) } } @@ -83,9 +82,7 @@ func TestStore_SeedIsValid(t *testing.T) { // decision (deployments/compose/settings ships the opt-in trial one). assert.Len(t, findings, 1, "findings: %s", findingStrings(findings)) assert.Contains(t, findingStrings(findings), "no policy") - _, id, req := s.DedupeFor("anything") - assert.Equal(t, "event_id", id) - assert.False(t, req) + assert.Equal(t, Dedupe{IDField: "event_id"}, s.DedupeFor("anything"), "retention 0: ids kept forever, as before retention existed") assert.Equal(t, ClickHouse{Addr: "localhost:9000", HTTPPort: 8123, HTTPScheme: "http", Database: "default", Username: "default", QueryTimeout: 30 * time.Second, Headers: map[string]string{}, MaxOpenConns: 10, MaxIdleConns: 5}, s.ClickHouse()) assert.Equal(t, Auth{JWKSURL: "", RoleClaim: "role"}, s.Auth()) assert.True(t, s.DLQFor("anything")) diff --git a/internal/settings/validate.go b/internal/settings/validate.go index 08575c953..76352e199 100644 --- a/internal/settings/validate.go +++ b/internal/settings/validate.go @@ -12,6 +12,7 @@ import ( "path/filepath" "slices" "strings" + "time" "github.com/Wave-RF/WaveHouse/internal/pipes" "github.com/Wave-RF/WaveHouse/internal/policy" @@ -450,6 +451,26 @@ func (v *validator) checkIDField(path string, val *string) { } } +// checkRetention rejects a dedupe retention that is not a duration, is +// negative, or is finite but shorter than MinDedupeRetention. The short one +// is refused rather than raised to the minimum, so the file never means +// something other than what it says. nil is the caller's concern, as for +// id_field. +func (v *validator) checkRetention(path string, val *string) { + if val == nil { + return + } + d, err := time.ParseDuration(*val) + switch { + case err != nil: + v.errorf(FileConfig, path, "must be a duration such as \"720h\", or \"0\" to keep ids forever, got %q", *val) + case d < 0: + v.errorf(FileConfig, path, "must not be negative, got %q", *val) + case d > 0 && d < MinDedupeRetention: + v.errorf(FileConfig, path, "%q is shorter than the ingest queue's %s duplicate window: an id re-sent after it expires but inside the window would be dropped by the queue while the client is told it was accepted — use at least %q, or \"0\" to keep ids forever", *val, MinDedupeRetention, MinDedupeRetention.String()) + } +} + // checkTableName rejects a per-table override key that could never match a // table: empty, carrying surrounding whitespace, or holding NUL. Shared by // the dedupe and dlq override maps. @@ -673,15 +694,20 @@ func (v *validator) parseConfig(data []byte) TenantConfig { if d.RequireID == nil { v.required("dedupe.require_id") } + if d.Retention == nil { + v.required("dedupe.retention") + } v.checkIDField("dedupe.id_field", d.IDField) + v.checkRetention("dedupe.retention", d.Retention) // Sorted iteration keeps finding order deterministic across runs. for _, table := range slices.Sorted(maps.Keys(d.Tables)) { td := d.Tables[table] path := "dedupe.tables." + table v.checkTableName("dedupe.tables", table) v.checkIDField(path+".id_field", td.IDField) - if td.IDField == nil && td.RequireID == nil { - v.warnf(FileConfig, path, "override sets nothing — remove it, or set id_field or require_id") + v.checkRetention(path+".retention", td.Retention) + if td.IDField == nil && td.RequireID == nil && td.Retention == nil { + v.warnf(FileConfig, path, "override sets nothing — remove it, or set id_field, require_id or retention") } } } diff --git a/internal/settings/validate_test.go b/internal/settings/validate_test.go index 4dae9c9d4..a6d7c5f73 100644 --- a/internal/settings/validate_test.go +++ b/internal/settings/validate_test.go @@ -269,13 +269,13 @@ func TestValidate_ContentRules(t *testing.T) { {"negative max rows", FileConfig, `{"query": {"default_max_rows": -1}}`, "must be >= 1"}, {"zero max rows", FileConfig, `{"query": {"default_max_rows": 0}}`, "must be >= 1"}, {"missing dedupe block", FileConfig, `{"dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "dedupe: required"}, - {"missing dlq block", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "dlq: required"}, + {"missing dlq block", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "dlq: required"}, {"missing dlq.enabled", FileConfig, `{"dlq": {"tables": {}}}`, "dlq.enabled: required"}, {"empty dlq override table name", FileConfig, `{"dlq": {"tables": {"": {"enabled": false}}}}`, "table name must not be empty"}, {"dlq override table whitespace", FileConfig, `{"dlq": {"tables": {"clicks ": {"enabled": false}}}}`, "surrounding whitespace"}, {"missing query.timestamp_bucket_seconds", FileConfig, `{"query": {"default_max_rows": 1}}`, "query.timestamp_bucket_seconds: required"}, {"negative timestamp bucket", FileConfig, `{"query": {"timestamp_bucket_seconds": -1}}`, "must be >= 0"}, - {"missing stream block", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "cors": {"allowed_origins": []}}`, "stream: required"}, + {"missing stream block", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "cors": {"allowed_origins": []}}`, "stream: required"}, {"missing stream.keepalive_interval", FileConfig, `{"stream": {"keepalive_buckets": 3, "gap_window_minutes": 15}}`, "stream.keepalive_interval: required"}, {"missing stream.keepalive_buckets", FileConfig, `{"stream": {"keepalive_interval": 30, "gap_window_minutes": 15}}`, "stream.keepalive_buckets: required"}, {"missing stream.gap_window_minutes", FileConfig, `{"stream": {"keepalive_interval": 30, "keepalive_buckets": 3}}`, "stream.gap_window_minutes: required"}, @@ -284,14 +284,21 @@ func TestValidate_ContentRules(t *testing.T) { {"negative gap window", FileConfig, `{"stream": {"gap_window_minutes": -1}}`, "stream.gap_window_minutes: must be >= 0"}, {"keepalive as a duration string", FileConfig, `{"stream": {"keepalive_interval": "30s"}}`, "keepalive_interval"}, {"missing dedupe.require_id", FileConfig, `{"dedupe": {"id_field": "event_id"}}`, "dedupe.require_id: required"}, - {"missing dedupe.enabled", FileConfig, `{"dedupe": {"id_field": "event_id", "require_id": false}}`, "dedupe.enabled: required"}, + {"missing dedupe.enabled", FileConfig, `{"dedupe": {"id_field": "event_id", "require_id": false, "retention": "0"}}`, "dedupe.enabled: required"}, + {"missing dedupe.retention", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}}`, "dedupe.retention: required"}, + {"dedupe.retention not a duration", FileConfig, configJSON(`{"dedupe": {"retention": "30d"}}`), `dedupe.retention: must be a duration such as "720h"`}, + {"dedupe.retention a number", FileConfig, configJSON(`{"dedupe": {"retention": 3600}}`), "retention"}, + {"dedupe.retention negative", FileConfig, configJSON(`{"dedupe": {"retention": "-1h"}}`), "dedupe.retention: must not be negative"}, + {"dedupe.retention under the duplicate window", FileConfig, configJSON(`{"dedupe": {"retention": "1m59s"}}`), `dedupe.retention: "1m59s" is shorter than the ingest queue's 2m0s duplicate window`}, + {"override retention under the duplicate window", FileConfig, configJSON(`{"dedupe": {"tables": {"clicks": {"retention": "30s"}}}}`), "dedupe.tables.clicks.retention: \"30s\" is shorter"}, + {"override retention not a duration", FileConfig, configJSON(`{"dedupe": {"tables": {"clicks": {"retention": "forever"}}}}`), "dedupe.tables.clicks.retention: must be a duration"}, {"missing query.default_max_rows", FileConfig, `{"query": {}}`, "query.default_max_rows: required"}, {"missing schema.refresh_interval", FileConfig, `{"schema": {}}`, "schema.refresh_interval: required"}, {"missing cors.allowed_origins", FileConfig, `{"cors": {}}`, "cors.allowed_origins: required"}, {"empty config document", FileConfig, `{}`, "cors: required"}, {"otel is boot config", FileConfig, `{"otel": {"enabled": true}}`, "unknown field"}, - {"missing clickhouse block", FileConfig, `{"auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "clickhouse: required"}, - {"missing auth block", FileConfig, `{"clickhouse": {"addr": "h:9000", "http_port": 8123, "http_scheme": "http", "database": "d", "username": "u", "query_timeout": 1}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "auth: required"}, + {"missing clickhouse block", FileConfig, `{"auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "clickhouse: required"}, + {"missing auth block", FileConfig, `{"clickhouse": {"addr": "h:9000", "http_port": 8123, "http_scheme": "http", "database": "d", "username": "u", "query_timeout": 1}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": 1, "timestamp_bucket_seconds": 0}, "schema": {"refresh_interval": 1}, "stream": {"keepalive_interval": 1, "keepalive_buckets": 1, "gap_window_minutes": 0}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": []}}`, "auth: required"}, {"missing clickhouse.addr", FileConfig, `{"clickhouse": {"http_port": 8123, "http_scheme": "http", "database": "d", "username": "u", "query_timeout": 1}}`, "clickhouse.addr: required"}, {"clickhouse.addr without port", FileConfig, `{"clickhouse": {"addr": "localhost"}}`, "must be host:port"}, {"clickhouse.http_port out of range", FileConfig, `{"clickhouse": {"http_port": 70000}}`, "clickhouse.http_port: must be in 1-65535"}, @@ -344,6 +351,28 @@ func TestValidate_ContentRules(t *testing.T) { } } +// A finite retention at or above the duplicate window is accepted, "0" (or +// any zero duration) is forever, and a table may keep ids longer or shorter +// than the tenant, or forever under a finite tenant retention. +func TestValidate_DedupeRetentionAccepted(t *testing.T) { + t.Parallel() + for _, patch := range []string{ + `{"dedupe": {"retention": "0"}}`, + `{"dedupe": {"retention": "0s"}}`, + `{"dedupe": {"retention": "2m"}}`, + `{"dedupe": {"retention": "720h", "tables": {"clicks": {"retention": "24h"}, "views": {"retention": "0"}}}}`, + } { + t.Run(patch, func(t *testing.T) { + t.Parallel() + files := validFiles() + files[FileConfig] = configJSON(patch) + doc, findings := ValidateDir(writeDir(t, files)) + require.NotNil(t, doc, "findings: %s", findingStrings(findings)) + assert.False(t, HasErrors(findings)) + }) + } +} + // TestValidate_ClickHouseTLSPathsAreNotOpened pins that the tls block is // checked for shape only: Validate is pure and also runs on the control // plane, so paths that exist nowhere still validate, and the values reach diff --git a/internal/testutil/mocks.go b/internal/testutil/mocks.go index 624fe0efd..b82974e7e 100644 --- a/internal/testutil/mocks.go +++ b/internal/testutil/mocks.go @@ -116,6 +116,7 @@ func (m *MockSubscriber) Close() error { return nil } type MockDeduplicator struct { mu sync.Mutex committed map[dedupe.Key]bool + retention map[dedupe.Key]time.Duration // each commit's retention pending map[dedupe.Key]string tokens int // Err, if set, fails Reserve — after ErrAfter calls have succeeded; @@ -132,7 +133,7 @@ type MockDeduplicator struct { var _ dedupe.Deduplicator = (*MockDeduplicator)(nil) func NewMockDeduplicator() *MockDeduplicator { - return &MockDeduplicator{committed: map[dedupe.Key]bool{}, pending: map[dedupe.Key]string{}} + return &MockDeduplicator{committed: map[dedupe.Key]bool{}, retention: map[dedupe.Key]time.Duration{}, pending: map[dedupe.Key]string{}} } // Reserve answers Duplicate for a key repeated in one call, as Managed does. @@ -163,7 +164,7 @@ func (m *MockDeduplicator) Reserve(_ context.Context, keys []dedupe.Key, _ time. return claims, nil } -func (m *MockDeduplicator) Commit(_ context.Context, claims []dedupe.Claim, _ time.Duration) error { +func (m *MockDeduplicator) Commit(_ context.Context, claims []dedupe.Claim, retention time.Duration) error { m.mu.Lock() defer m.mu.Unlock() m.Commits++ @@ -173,6 +174,7 @@ func (m *MockDeduplicator) Commit(_ context.Context, claims []dedupe.Claim, _ ti for _, c := range claims { if c.Status == dedupe.Claimed { m.committed[c.Key] = true + m.retention[c.Key] = retention delete(m.pending, c.Key) } } @@ -209,6 +211,13 @@ func (m *MockDeduplicator) Committed(k dedupe.Key) bool { return m.committed[k] } +// Retention is the retention k was last committed with. +func (m *MockDeduplicator) Retention(k dedupe.Key) time.Duration { + m.mu.Lock() + defer m.mu.Unlock() + return m.retention[k] +} + // Pending reports whether k is claimed and neither committed nor released. func (m *MockDeduplicator) Pending(k dedupe.Key) bool { m.mu.Lock() diff --git a/tests/e2e/fixtures/settings/config.json b/tests/e2e/fixtures/settings/config.json index a0d15cfce..a8d7c7367 100644 --- a/tests/e2e/fixtures/settings/config.json +++ b/tests/e2e/fixtures/settings/config.json @@ -26,6 +26,7 @@ "enabled": true, "id_field": "event_id", "require_id": false, + "retention": "0", "tables": {} }, "dlq": { From 6c74b65969b26e6cc5bc94ac6957f001ffbed8cb Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Fri, 25 Sep 2026 04:16:23 -0400 Subject: [PATCH 02/11] test(dedupe): pin that a commit mid-sweep-chunk survives; docs wording Adds a sweep hook so a Commit can race into the gap between a chunk's read and delete; the test fails (3/3) with commitMu removed. Docs: the sweep follows the shared instance, reads (not deletes) 1,024 keys per chunk, and "0" is the one unitless retention. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01EJr5tY4WQUy2sc4MbW67vL --- CHANGELOG.md | 2 +- docs/src/content/docs/durability.md | 2 +- docs/src/content/docs/settings-directory.mdx | 2 +- internal/dedupe/embedded.go | 3 ++ internal/dedupe/sweep.go | 3 ++ internal/dedupe/sweep_test.go | 35 ++++++++++++++++++++ 6 files changed, 44 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f47f81ba6..03affb5f8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,7 +29,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Settings-directory validation — `wavehouse validate [dir]`** (`internal/settings/` (new: `settings.go`, `validate.go`, `decode.go`, `finding.go`, + tests), `cmd/wavehouse/validate.go` (new, + tests), `cmd/wavehouse/main.go`): first piece of the file-based control plane (settings live in a directory of JSON documents — `roles.json`, `policies.json`, `pipes.json`, `config.json` — that a running instance will hot-reload; this change is validation-only — boot loading and reload wiring land separately). `settings.Validate(dir)` is the single gate every consumer of the directory runs: deliberately pure (no network, no ClickHouse — table/column existence stays with schema discovery, per Bring-Your-Own-Schema), and it collects **all** findings in one pass instead of failing on the first. Checks, layered: the directory holds exactly the four files (a missing file is an error — an empty document is `{}`, so absence always means deletion or a wrong path; any unexpected entry — file or directory — is an error so a typoed `polices.json` or a stray backup can't be silently ignored; dot-prefixed entries are the one carve-out, since erroring on vim swap files or the `..data` machinery Kubernetes ConfigMap mounts publish through would break hand editing and the cloud fan-out's mount pattern alike); strict JSON syntax (unknown fields rejected — the JSON form of the retired-config-key trap; empty/truncated files rejected, never read as an empty document; a leading UTF-8 byte order mark named as such instead of surfacing as a cryptic invalid-character error; a directory, unreadable file, or non-regular file (a FIFO would hang the read forever waiting for a writer; a stat gate rejects it — following symlinks, so Kubernetes ConfigMap mounts' symlink layout still passes) squatting on a settings filename named as the one real problem, not double-reported as "missing"; a top-level `null` rejected — the one well-formed document that decodes into a zero value without error, so it would silently read as "no settings"; trailing content rejected; duplicated object keys detected by a token-level pass, since `encoding/json` silently keeps the last copy); per-file shape rules (role names non-empty/unique, pipe names/SQL/param types, `config.json` bounds mirroring boot-config validation — its sections are the *tenant-owned* behavioral tunables (dedupe id_field/require_id plus per-table overrides under `dedupe.tables` — each entry overrides only the fields it names, resolving table → global → compiled default per field, so the effective id_field can never be empty — an explicit empty, whitespace-only, or whitespace-padded id_field is rejected at both levels, since an exact-match JSON key lookup would silently miss every row ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s shape, unblocked by the file design since table names are runtime-resolved like policy grants); query default_max_rows, schema refresh_interval, CORS origins); platform-owned knobs like the SSE keepalives deliberately stay boot config); and cross-file referential integrity (every role a policy grant, `default_role`/`admin_role`, or pipe allowlist references must be declared in `roles.json`; an empty role string in a grant or allowlist is named as such — it matches no request and authorizes nobody). Warnings don't invalidate: a grant scoping the admin role (an unconditional bypass — dead config), `default_role` = admin, and a `default` on a required pipe parameter are flagged but legal. An empty `policies.json` means no policy — fail closed, matching deleted-policy semantics — and draws a warning naming the total lockout, so it announces itself at validation time instead of one 403 at a time. The CLI (`cmd/wavehouse/validate.go`, following the `health` subcommand pattern) takes the directory as an argument or from `WH_SETTINGS_DIR`, prints findings, and exits 0/1/2 (valid/invalid/usage) so CI and operators can gate config changes before they reach a running instance. The dispatch in `main.go` also grows `help` and `version` subcommands, and an unknown command is now a usage error instead of silently falling through and starting the server (`wavehouse validat` booting a listener is not a typo anyone wants); each subcommand parses its arguments with a stdlib `flag.FlagSet`, so `wavehouse -h` prints command-specific help and a stray flag or argument is a usage error rather than being silently swallowed. `WH_SETTINGS_DIR` has a single authority: `config.EnvSettingsDir`, with a reflection test pinning the `settings.dir` struct tag to it. The directory's location joins boot config as `settings.dir` (`WH_SETTINGS_DIR`; `internal/config/config.go`, `config.yaml`, `docs/src/content/docs/configuration.mdx`) — boot-tier by necessity, since it's the pointer the reload machinery follows; no default, same silent-misconfiguration reasoning as `policy.file_path`. - **Docs-site analytics for search, code copies, 404s, docs section, and live-demo connectivity** (`docs/src/components/DocsTracking.astro` (new), `docs/src/components/{PostHog,Footer,LiveDemo}.astro`): the site tracked its own CTAs but nothing a reader did on the way to one, so the questions that decide what to write next — what people search for and *don't* find, which snippets get copied, which dead links keep getting followed — had no data behind them. `docs_search` fires a second after the query settles rather than once per keystroke, carrying `query` and `result_count` read off Pagefind's own results message (the rendered list is capped at its page size, so counting the DOM would under-report); `result_count: 0` is the event worth having. `code_copied` (`page`, `language`) watches Expressive Code's copy buttons from the document rather than re-binding every code block on every navigation — the hero's install chip is not an EC block and keeps its own `hero_install_copied`. `docs_404` (`path`, `referrer`) turns broken inbound links into a list instead of a hunch. A `doc_section` property (the first path segment, `home` for `/`) puts every event in a docs area without each tracker carrying its own copy; it's stamped at capture time by a `before_send` hook in `posthog.init()` rather than `register()`, because a queued `register()` replays only after init has already captured the first hard-load `$pageview` — which would then carry the previous visit's persisted value — and `history_change` navigations update the URL before capture fires, so reading `location` in the hook is always current. `live_demo_connected` fires once per mount when the hero's SSE feed comes up rather than on its first row — named for what it measures (the demo backend answered), since a quiet minute on the repo is not a disengaged reader. The three site-wide trackers share one new `DocsTracking.astro` rendered from the footer (like `MermaidZoom` / `ScrollHints`) and delegate from `document`, since Pagefind, Expressive Code, and the 404 route all own their own markup — some of it created after page load. -- **Dedupe retention per tenant and table, and a sweep that deletes expired ids** (`internal/settings/{settings,validate,store}.go` (+ tests), `internal/settings/seed/config.json`, `internal/dedupe/{embedded,sweep}.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/testutil/mocks.go`, `deployments/compose/settings/config.json`, `tests/e2e/fixtures/settings/config.json`, `config.yaml`, `docs/src/content/docs/{settings-directory.mdx,deployment,durability,architecture}.md`, `AGENTS.md`): [#220](https://github.com/Wave-RF/WaveHouse/issues/220). `config.json` gains **`dedupe.retention`, a required key**, overridable per table in `dedupe.tables.
.retention`: how long a committed id stays a duplicate, as a duration string (`"720h"`), or `"0"` to keep it forever, which is the seed value and the behaviour before this release. **Every existing `config.json` must add it**; `"retention": "0"` changes nothing. A finite retention below `"2m"`, the ingest queue's duplicate window, is refused rather than raised to the minimum: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and dropped by the queue as a copy while the client was told it was accepted. It is hot-reloadable and read per record like `id_field`; a change applies to ids committed after it, and a reload landing mid-window commits each record with the retention it was prepared under. The embedded Pebble store already treated an expired id as new; it now also deletes expired keys, and the version-0 keys the key-layout change left behind, in a background sweep that starts a minute after the instance opens and repeats hourly. It deletes 1,024 keys per chunk without fsync, under a lock `Commit` also takes, so an id committed again after the sweep read it is never deleted. New metric `wavehouse_dedupe_swept_keys_total{reason="expired"|"version_0"}`. `settings.Store.DedupeFor` now returns a `settings.Dedupe` struct rather than three values. +- **Dedupe retention per tenant and table, and a sweep that deletes expired ids** (`internal/settings/{settings,validate,store}.go` (+ tests), `internal/settings/seed/config.json`, `internal/dedupe/{embedded,sweep}.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/testutil/mocks.go`, `deployments/compose/settings/config.json`, `tests/e2e/fixtures/settings/config.json`, `config.yaml`, `docs/src/content/docs/{settings-directory.mdx,deployment,durability,architecture}.md`, `AGENTS.md`): [#220](https://github.com/Wave-RF/WaveHouse/issues/220). `config.json` gains **`dedupe.retention`, a required key**, overridable per table in `dedupe.tables.
.retention`: how long a committed id stays a duplicate, as a duration string (`"720h"`), or `"0"` to keep it forever, which is the seed value and the behaviour before this release. **Every existing `config.json` must add it**; `"retention": "0"` changes nothing. A finite retention below `"2m"`, the ingest queue's duplicate window, is refused rather than raised to the minimum: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and dropped by the queue as a copy while the client was told it was accepted. It is hot-reloadable and read per record like `id_field`; a change applies to ids committed after it, and a reload landing mid-window commits each record with the retention it was prepared under. The embedded Pebble store already treated an expired id as new; it now also deletes expired keys, and the version-0 keys the key-layout change left behind, in a background sweep that starts a minute after the instance opens and repeats hourly. It reads 1,024 keys per chunk and deletes the expired and version-0 ones, without fsync, under a lock `Commit` also takes, so an id committed again after the sweep read it is never deleted. New metric `wavehouse_dedupe_swept_keys_total{reason="expired"|"version_0"}`. `settings.Store.DedupeFor` now returns a `settings.Dedupe` struct rather than three values. ### Changed diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index a0a62952c..2ad53395b 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -62,7 +62,7 @@ The tell for a commit-cadence problem (ZFS-without-SLOG, noisy-neighbor VM host) With [deduplication](/settings-directory#deduplication) on, a `200` also means the records' ids were committed to the dedupe store, or, if that commit failed, that the failure was counted by `wavehouse_dedupe_commit_failed_total` and the ids lapse with their lease. On the embedded Pebble store that commit is an `fsync` of its own. It is taken once per window of up to 256 records of a request, after the window's publishes, rather than once per record: a 1,000-record batch costs four dedupe syncs, not a thousand. Measured with `BenchmarkIngest_DedupBatchOnPebble` on an Apple M4 Pro under a load average near 30, with the queue stubbed out so only the dedupe store touched disk, the dedupe work for that batch took 24 ms windowed against 5.7 s one record at a time; the JetStream publishes' own fsyncs come on top. A single-record request still pays one sync for its publish and one for its commit. -With a finite `dedupe.retention`, expired ids are deleted by a background sweep, an hour apart. Its deletes are not fsynced (a delete lost to a crash is redone by the next pass), so it adds no sync to the ingest path; it reads and deletes 1,024 keys at a time, and a commit that arrives mid-chunk waits for that chunk. An expired id is already treated as new by the next claim of it, sweep or no sweep, so retention never depends on the sweep having run. +With a finite `dedupe.retention`, expired ids are deleted by a background sweep, an hour apart. Its deletes are not fsynced (a delete lost to a crash is redone by the next pass), so it adds no sync to the ingest path; it reads 1,024 keys at a time, deleting the expired ones, and a commit that arrives mid-chunk waits for that chunk. An expired id is already treated as new by the next claim of it, sweep or no sweep, so retention never depends on the sweep having run. A publish can also fail after JetStream stored the event (a timeout on the ack). The record's id is then left to lapse with its 30-second dedupe lease rather than given back, and every deduped record is published under an idempotency key derived from its tenant, table and id, which each tenant's ingest stream remembers for two minutes after the first publish. A retry after the lease but inside those two minutes is therefore dropped by the stream rather than stored twice; one later than that is stored again. The 30-second lease sits well inside those two minutes, so a prompt retry is covered. For the same reason a finite `dedupe.retention` must be at least those two minutes: an id re-sent after a shorter retention ended would be claimed again, then dropped by the stream as a copy while the client was told it was accepted. Settings validation refuses one below it. diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 126f35dd1..b9d301626 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -190,7 +190,7 @@ Every dedupe knob lives here — there are no boot-config keys for it. The switc - `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens or closes this tenant's store in the embedded Pebble instance at `/pebble`, so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`503 dedupe store unavailable`, `Retry-After: 5`) until the next reload or restart — the files asked for dedupe, so publishing un-deduped is not a fallback. At boot a failed open refuses to start, like every other store. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) every tenant's seen ids live in that one instance, each key led by its tenant and table, and it is open while any tenant's switch is on: each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and keeps its seen ids for the folder that restores it; and if that instance fails to open, at boot or on reload, every tenant with dedupe on fails closed — its ingest answers `503 dedupe store unavailable` (`Retry-After: 5`) until a reload opens it — while the tenants with dedupe off carry on. - `dedupe.id_field` (seed default `event_id`) — JSON field name in the ingest body used as the dedup key. An id is a duplicate only within its own tenant and table: the same value in two tables is two ids. An id longer than 1,024 bytes is stored as its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. An id is committed only after its record is published; if that commit fails (counted by `wavehouse_dedupe_commit_failed_total`, which should stay at zero), the record is still answered `ok` and the id lapses with its 30-second lease: a retry of it before then answers in-flight, one inside the ingest queue's two-minute duplicate window is dropped there by its idempotency key, and one after that is stored again. - `dedupe.require_id` (seed default `false`) — controls what happens to a row missing `id_field`, or carrying it as `null` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. -- `dedupe.retention` (seed default `"0"`) — how long a committed id stays a duplicate, as a Go duration string: `"24h"`, `"720h"` (30 days), `"90m"`. There is no day unit. `"0"` keeps every id forever, which was the only behavior before this key existed. Once an id's retention has ended, the next record carrying it is published as new, and a background sweep deletes the expired id from the store: first a minute after the store opens, then hourly, counted by `wavehouse_dedupe_swept_keys_total{reason="expired"}`. A finite retention must be at least `"2m"`, the ingest queue's duplicate window: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and then dropped by the queue as a copy, while the client was told it was accepted. A retention below that is refused, not raised to the minimum; so are a negative value and anything that is not a duration, such as `"30d"` or a bare number. Hot-reloadable: a change applies to ids committed after the reload, and an id already committed keeps the expiry it was stored with. +- `dedupe.retention` (seed default `"0"`) — how long a committed id stays a duplicate, as a Go duration string: `"24h"`, `"720h"` (30 days), `"90m"`. There is no day unit. `"0"` keeps every id forever, which was the only behavior before this key existed. Once an id's retention has ended, the next record carrying it is published as new, and a background sweep over the shared Pebble instance deletes the expired id: first about a minute after the instance opens (when the first tenant switches dedupe on), then hourly while any tenant keeps it on, counted by `wavehouse_dedupe_swept_keys_total{reason="expired"}`. A finite retention must be at least `"2m"`, the ingest queue's duplicate window: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and then dropped by the queue as a copy, while the client was told it was accepted. A retention below that is refused, not raised to the minimum; so are a negative value and anything that is not a duration, such as `"30d"`, a number with no unit (`"300"` needs one: `"300s"`; `"0"` is the one exception), or a JSON number rather than a string. Hot-reloadable: a change applies to ids committed after the reload, and an id already committed keeps the expiry it was stored with. - `dedupe.tables.
.{id_field, require_id, retention}` — per-table overrides; each entry overrides only the fields it names and inherits the rest. A table can keep ids for a shorter time than its tenant, or for longer, or forever (`"retention": "0"`) under a finite tenant retention. ## ClickHouse diff --git a/internal/dedupe/embedded.go b/internal/dedupe/embedded.go index 5186a7380..1a620412d 100644 --- a/internal/dedupe/embedded.go +++ b/internal/dedupe/embedded.go @@ -47,6 +47,9 @@ type Embedded struct { // readHook, when set, runs before each Pebble read in Reserve; a test // makes it fail to exercise Reserve's all-or-nothing error path. readHook func() error + // sweepHook, when set, runs in a sweep chunk between reading its keys and + // deleting them; a test races a Commit into that gap. + sweepHook func() } // NewEmbedded returns the embedded implementation under dataDir. Nothing is diff --git a/internal/dedupe/sweep.go b/internal/dedupe/sweep.go index e536c5edc..396d64556 100644 --- a/internal/dedupe/sweep.go +++ b/internal/dedupe/sweep.go @@ -135,6 +135,9 @@ func (e *Embedded) sweepChunk(ctx context.Context, db *pebble.DB, from []byte, r if err := it.Close(); err != nil { return nil, fmt.Errorf("dedupe sweep: %w", err) } + if e.sweepHook != nil { + e.sweepHook() + } // NoSync: a delete lost to a crash is redone by the next pass. if err := b.Commit(pebble.NoSync); err != nil { return nil, fmt.Errorf("dedupe sweep: %w", err) diff --git a/internal/dedupe/sweep_test.go b/internal/dedupe/sweep_test.go index 44579dcd8..0589c1030 100644 --- a/internal/dedupe/sweep_test.go +++ b/internal/dedupe/sweep_test.go @@ -97,6 +97,41 @@ func TestEmbedded_SweepDeletesExpiredAndVersionZeroKeys(t *testing.T) { assert.Equal(t, sweepResult{}, res, "a second pass finds nothing") } +// A Commit that arrives while a sweep chunk has read an expired key but not +// yet deleted it waits for the chunk, so the new commit is never deleted with +// the old value. Without the lock the Commit lands in the gap and the sweep +// then deletes it; the wait below only ever lets that pass, never fail. +func TestEmbedded_SweepNeverDeletesACommitLandingMidChunk(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + clock := newStepClock() + SetClock(e, clock.now) + m := switchedOn(t, e, "acme") + commitIDs(t, m, time.Hour, "e1") + clock.advance(2 * time.Hour) + + claims, err := m.Reserve(context.Background(), []Key{{Table: "events", ID: "e1"}}, DefaultLease) + require.NoError(t, err) + require.Equal(t, Claimed, claims[0].Status, "expired: claimable again") + done := make(chan error, 1) + e.sweepHook = func() { + go func() { done <- m.Commit(context.Background(), claims, time.Hour) }() + select { + case <-done: + done <- nil + case <-time.After(50 * time.Millisecond): + } + } + res, err := e.sweep(context.Background(), e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{Expired: 1}, res) + require.NoError(t, <-done) + + dup, err := mark(context.Background(), m, "e1") + require.NoError(t, err) + assert.True(t, dup, "the commit made mid-chunk survived the sweep") +} + // A retention is honoured on read before any sweep has run: the key is a // duplicate until the retention ends and claimable from that instant. func TestEmbedded_RetentionHonouredOnRead(t *testing.T) { From 7b9595e35cd9a5d2aca3ab5ef50cd7527839f5ec Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Fri, 25 Sep 2026 12:25:03 -0400 Subject: [PATCH 03/11] feat(settings): a missing dedupe.retention keeps ids forever MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit dedupe.retention was required in every config.json. It is now optional: missing means "0" (forever) at the tenant level, and a table override without it inherits the tenant's, as for the other override fields. A present value is validated as before — unparseable, negative, or finite below the queue's duplicate window is refused — so an existing directory needs no change to upgrade. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01EJr5tY4WQUy2sc4MbW67vL --- CHANGELOG.md | 2 +- docs/src/content/docs/deployment.md | 2 +- docs/src/content/docs/settings-directory.mdx | 8 +++---- internal/settings/settings.go | 11 +++++---- internal/settings/store.go | 5 +++- internal/settings/store_test.go | 17 +++++++++++++ internal/settings/validate.go | 7 ++---- internal/settings/validate_test.go | 25 +++++++++++++++++++- 8 files changed, 59 insertions(+), 18 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 03affb5f8..ffa7cb6fb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,7 +29,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Settings-directory validation — `wavehouse validate [dir]`** (`internal/settings/` (new: `settings.go`, `validate.go`, `decode.go`, `finding.go`, + tests), `cmd/wavehouse/validate.go` (new, + tests), `cmd/wavehouse/main.go`): first piece of the file-based control plane (settings live in a directory of JSON documents — `roles.json`, `policies.json`, `pipes.json`, `config.json` — that a running instance will hot-reload; this change is validation-only — boot loading and reload wiring land separately). `settings.Validate(dir)` is the single gate every consumer of the directory runs: deliberately pure (no network, no ClickHouse — table/column existence stays with schema discovery, per Bring-Your-Own-Schema), and it collects **all** findings in one pass instead of failing on the first. Checks, layered: the directory holds exactly the four files (a missing file is an error — an empty document is `{}`, so absence always means deletion or a wrong path; any unexpected entry — file or directory — is an error so a typoed `polices.json` or a stray backup can't be silently ignored; dot-prefixed entries are the one carve-out, since erroring on vim swap files or the `..data` machinery Kubernetes ConfigMap mounts publish through would break hand editing and the cloud fan-out's mount pattern alike); strict JSON syntax (unknown fields rejected — the JSON form of the retired-config-key trap; empty/truncated files rejected, never read as an empty document; a leading UTF-8 byte order mark named as such instead of surfacing as a cryptic invalid-character error; a directory, unreadable file, or non-regular file (a FIFO would hang the read forever waiting for a writer; a stat gate rejects it — following symlinks, so Kubernetes ConfigMap mounts' symlink layout still passes) squatting on a settings filename named as the one real problem, not double-reported as "missing"; a top-level `null` rejected — the one well-formed document that decodes into a zero value without error, so it would silently read as "no settings"; trailing content rejected; duplicated object keys detected by a token-level pass, since `encoding/json` silently keeps the last copy); per-file shape rules (role names non-empty/unique, pipe names/SQL/param types, `config.json` bounds mirroring boot-config validation — its sections are the *tenant-owned* behavioral tunables (dedupe id_field/require_id plus per-table overrides under `dedupe.tables` — each entry overrides only the fields it names, resolving table → global → compiled default per field, so the effective id_field can never be empty — an explicit empty, whitespace-only, or whitespace-padded id_field is rejected at both levels, since an exact-match JSON key lookup would silently miss every row ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s shape, unblocked by the file design since table names are runtime-resolved like policy grants); query default_max_rows, schema refresh_interval, CORS origins); platform-owned knobs like the SSE keepalives deliberately stay boot config); and cross-file referential integrity (every role a policy grant, `default_role`/`admin_role`, or pipe allowlist references must be declared in `roles.json`; an empty role string in a grant or allowlist is named as such — it matches no request and authorizes nobody). Warnings don't invalidate: a grant scoping the admin role (an unconditional bypass — dead config), `default_role` = admin, and a `default` on a required pipe parameter are flagged but legal. An empty `policies.json` means no policy — fail closed, matching deleted-policy semantics — and draws a warning naming the total lockout, so it announces itself at validation time instead of one 403 at a time. The CLI (`cmd/wavehouse/validate.go`, following the `health` subcommand pattern) takes the directory as an argument or from `WH_SETTINGS_DIR`, prints findings, and exits 0/1/2 (valid/invalid/usage) so CI and operators can gate config changes before they reach a running instance. The dispatch in `main.go` also grows `help` and `version` subcommands, and an unknown command is now a usage error instead of silently falling through and starting the server (`wavehouse validat` booting a listener is not a typo anyone wants); each subcommand parses its arguments with a stdlib `flag.FlagSet`, so `wavehouse -h` prints command-specific help and a stray flag or argument is a usage error rather than being silently swallowed. `WH_SETTINGS_DIR` has a single authority: `config.EnvSettingsDir`, with a reflection test pinning the `settings.dir` struct tag to it. The directory's location joins boot config as `settings.dir` (`WH_SETTINGS_DIR`; `internal/config/config.go`, `config.yaml`, `docs/src/content/docs/configuration.mdx`) — boot-tier by necessity, since it's the pointer the reload machinery follows; no default, same silent-misconfiguration reasoning as `policy.file_path`. - **Docs-site analytics for search, code copies, 404s, docs section, and live-demo connectivity** (`docs/src/components/DocsTracking.astro` (new), `docs/src/components/{PostHog,Footer,LiveDemo}.astro`): the site tracked its own CTAs but nothing a reader did on the way to one, so the questions that decide what to write next — what people search for and *don't* find, which snippets get copied, which dead links keep getting followed — had no data behind them. `docs_search` fires a second after the query settles rather than once per keystroke, carrying `query` and `result_count` read off Pagefind's own results message (the rendered list is capped at its page size, so counting the DOM would under-report); `result_count: 0` is the event worth having. `code_copied` (`page`, `language`) watches Expressive Code's copy buttons from the document rather than re-binding every code block on every navigation — the hero's install chip is not an EC block and keeps its own `hero_install_copied`. `docs_404` (`path`, `referrer`) turns broken inbound links into a list instead of a hunch. A `doc_section` property (the first path segment, `home` for `/`) puts every event in a docs area without each tracker carrying its own copy; it's stamped at capture time by a `before_send` hook in `posthog.init()` rather than `register()`, because a queued `register()` replays only after init has already captured the first hard-load `$pageview` — which would then carry the previous visit's persisted value — and `history_change` navigations update the URL before capture fires, so reading `location` in the hook is always current. `live_demo_connected` fires once per mount when the hero's SSE feed comes up rather than on its first row — named for what it measures (the demo backend answered), since a quiet minute on the repo is not a disengaged reader. The three site-wide trackers share one new `DocsTracking.astro` rendered from the footer (like `MermaidZoom` / `ScrollHints`) and delegate from `document`, since Pagefind, Expressive Code, and the 404 route all own their own markup — some of it created after page load. -- **Dedupe retention per tenant and table, and a sweep that deletes expired ids** (`internal/settings/{settings,validate,store}.go` (+ tests), `internal/settings/seed/config.json`, `internal/dedupe/{embedded,sweep}.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/testutil/mocks.go`, `deployments/compose/settings/config.json`, `tests/e2e/fixtures/settings/config.json`, `config.yaml`, `docs/src/content/docs/{settings-directory.mdx,deployment,durability,architecture}.md`, `AGENTS.md`): [#220](https://github.com/Wave-RF/WaveHouse/issues/220). `config.json` gains **`dedupe.retention`, a required key**, overridable per table in `dedupe.tables.
.retention`: how long a committed id stays a duplicate, as a duration string (`"720h"`), or `"0"` to keep it forever, which is the seed value and the behaviour before this release. **Every existing `config.json` must add it**; `"retention": "0"` changes nothing. A finite retention below `"2m"`, the ingest queue's duplicate window, is refused rather than raised to the minimum: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and dropped by the queue as a copy while the client was told it was accepted. It is hot-reloadable and read per record like `id_field`; a change applies to ids committed after it, and a reload landing mid-window commits each record with the retention it was prepared under. The embedded Pebble store already treated an expired id as new; it now also deletes expired keys, and the version-0 keys the key-layout change left behind, in a background sweep that starts a minute after the instance opens and repeats hourly. It reads 1,024 keys per chunk and deletes the expired and version-0 ones, without fsync, under a lock `Commit` also takes, so an id committed again after the sweep read it is never deleted. New metric `wavehouse_dedupe_swept_keys_total{reason="expired"|"version_0"}`. `settings.Store.DedupeFor` now returns a `settings.Dedupe` struct rather than three values. +- **Dedupe retention per tenant and table, and a sweep that deletes expired ids** (`internal/settings/{settings,validate,store}.go` (+ tests), `internal/settings/seed/config.json`, `internal/dedupe/{embedded,sweep}.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/testutil/mocks.go`, `deployments/compose/settings/config.json`, `tests/e2e/fixtures/settings/config.json`, `config.yaml`, `docs/src/content/docs/{settings-directory.mdx,deployment,durability,architecture}.md`, `AGENTS.md`): [#220](https://github.com/Wave-RF/WaveHouse/issues/220). `config.json` gains an optional **`dedupe.retention`** key, overridable per table in `dedupe.tables.
.retention`: how long a committed id stays a duplicate, as a duration string (`"720h"`), or `"0"` to keep it forever, which is the seed value and the behaviour before this release. A `config.json` without the key keeps ids forever, and a table override without one inherits the tenant's, so an existing directory needs no change. A finite retention below `"2m"`, the ingest queue's duplicate window, is refused rather than raised to the minimum: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and dropped by the queue as a copy while the client was told it was accepted. It is hot-reloadable and read per record like `id_field`; a change applies to ids committed after it, and a reload landing mid-window commits each record with the retention it was prepared under. The embedded Pebble store already treated an expired id as new; it now also deletes expired keys, and the version-0 keys the key-layout change left behind, in a background sweep that starts a minute after the instance opens and repeats hourly. It reads 1,024 keys per chunk and deletes the expired and version-0 ones, without fsync, under a lock `Commit` also takes, so an id committed again after the sweep read it is never deleted. New metric `wavehouse_dedupe_swept_keys_total{reason="expired"|"version_0"}`. `settings.Store.DedupeFor` now returns a `settings.Dedupe` struct rather than three values. ### Changed diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 86347f444..ca23f0fe6 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -421,7 +421,7 @@ WaveHouse discovers this schema on startup and refreshes it every `schema.refres The dedupe key now carries the table as well as the tenant ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)), so **an id deduped before the upgrade is not recognized after it**: a record carrying it is accepted once more. Nothing is migrated. The old keys are never read, and the dedupe sweep deletes them: its first pass runs about a minute after the instance opens, and `wavehouse_dedupe_swept_keys_total{reason="version_0"}` counts them ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)). Pebble returns their disk space as it compacts, not at once. Only a tenant with `dedupe.enabled` on is affected, and only by a record sent both before and after the upgrade — typically a producer retrying across the restart. To avoid duplicate rows, let retrying producers finish, or pause them, before upgrading. -The same release adds **`dedupe.retention`, a required key**: every `config.json`, each tenant's folder included, must state it or the directory is refused (at boot) or not adopted (on reload). `"retention": "0"` keeps every id forever, as before; see [Deduplication](/settings-directory#deduplication) for a finite one. +The same release adds an optional **`dedupe.retention`** key. No upgrade step is needed: a `config.json` without it keeps every id forever, as before. See [Deduplication](/settings-directory#deduplication) for a finite one. ## Upgrading across the v2 ingest envelope diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index b9d301626..55a8e0d9c 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -99,7 +99,7 @@ Pipes are read per request, so a reload changes what the next `GET /v1/pipes/{na ## `config.json` keys -The tenant tunables. Every key is required (a missing one is a validation error) except the per-table overrides; the "Seed" column is what `wavehouse bootstrap` writes: +The tenant tunables. Every key is required (a missing one is a validation error) except `dedupe.retention` and the per-table overrides; the "Seed" column is what `wavehouse bootstrap` writes: | Key | Seed | Description | | --- | ---- | ----------- | @@ -123,7 +123,7 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `dedupe.enabled` | `false` | Turn deduplication on; a reload opens or closes this tenant's store — see [Deduplication](#deduplication). | | `dedupe.id_field` | `event_id` | Dedup key field — see [Deduplication](#deduplication). | | `dedupe.require_id` | `false` | Reject rows missing the id field — see [Deduplication](#deduplication). | -| `dedupe.retention` | `"0"` | How long a committed id stays a duplicate, as a duration (`"720h"`); `"0"` keeps it forever — see [Deduplication](#deduplication). | +| `dedupe.retention` | `"0"` | How long a committed id stays a duplicate, as a duration (`"720h"`); `"0"`, or leaving the key out, keeps it forever — see [Deduplication](#deduplication). | | `dedupe.tables.
.{id_field, require_id, retention}` | `{}` | Optional per-table overrides; each entry overrides only the fields it names and inherits the rest. | | `dlq.enabled` | `true` | Park poison rows — those that still fail after row-by-row isolation, and every row of a batch whose tenant has no ClickHouse connection — on the tenant's dead-letter stream (`DLQ_{tenant}`) (`false`: leave them unacked for redelivery — except an envelope the worker cannot read, which is dropped and counted) — see [Dead Letter Queue](#dead-letter-queue). | | `dlq.tables.
.enabled` | `{}` | Optional per-table override of the switch. | @@ -190,8 +190,8 @@ Every dedupe knob lives here — there are no boot-config keys for it. The switc - `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens or closes this tenant's store in the embedded Pebble instance at `/pebble`, so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`503 dedupe store unavailable`, `Retry-After: 5`) until the next reload or restart — the files asked for dedupe, so publishing un-deduped is not a fallback. At boot a failed open refuses to start, like every other store. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) every tenant's seen ids live in that one instance, each key led by its tenant and table, and it is open while any tenant's switch is on: each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and keeps its seen ids for the folder that restores it; and if that instance fails to open, at boot or on reload, every tenant with dedupe on fails closed — its ingest answers `503 dedupe store unavailable` (`Retry-After: 5`) until a reload opens it — while the tenants with dedupe off carry on. - `dedupe.id_field` (seed default `event_id`) — JSON field name in the ingest body used as the dedup key. An id is a duplicate only within its own tenant and table: the same value in two tables is two ids. An id longer than 1,024 bytes is stored as its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. An id is committed only after its record is published; if that commit fails (counted by `wavehouse_dedupe_commit_failed_total`, which should stay at zero), the record is still answered `ok` and the id lapses with its 30-second lease: a retry of it before then answers in-flight, one inside the ingest queue's two-minute duplicate window is dropped there by its idempotency key, and one after that is stored again. - `dedupe.require_id` (seed default `false`) — controls what happens to a row missing `id_field`, or carrying it as `null` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. -- `dedupe.retention` (seed default `"0"`) — how long a committed id stays a duplicate, as a Go duration string: `"24h"`, `"720h"` (30 days), `"90m"`. There is no day unit. `"0"` keeps every id forever, which was the only behavior before this key existed. Once an id's retention has ended, the next record carrying it is published as new, and a background sweep over the shared Pebble instance deletes the expired id: first about a minute after the instance opens (when the first tenant switches dedupe on), then hourly while any tenant keeps it on, counted by `wavehouse_dedupe_swept_keys_total{reason="expired"}`. A finite retention must be at least `"2m"`, the ingest queue's duplicate window: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and then dropped by the queue as a copy, while the client was told it was accepted. A retention below that is refused, not raised to the minimum; so are a negative value and anything that is not a duration, such as `"30d"`, a number with no unit (`"300"` needs one: `"300s"`; `"0"` is the one exception), or a JSON number rather than a string. Hot-reloadable: a change applies to ids committed after the reload, and an id already committed keeps the expiry it was stored with. -- `dedupe.tables.
.{id_field, require_id, retention}` — per-table overrides; each entry overrides only the fields it names and inherits the rest. A table can keep ids for a shorter time than its tenant, or for longer, or forever (`"retention": "0"`) under a finite tenant retention. +- `dedupe.retention` (optional; seed default `"0"`) — how long a committed id stays a duplicate, as a Go duration string: `"24h"`, `"720h"` (30 days), `"90m"`. There is no day unit. `"0"` keeps every id forever, which was the only behavior before this key existed, and a `config.json` without the key means the same. Once an id's retention has ended, the next record carrying it is published as new, and a background sweep over the shared Pebble instance deletes the expired id: first about a minute after the instance opens (when the first tenant switches dedupe on), then hourly while any tenant keeps it on, counted by `wavehouse_dedupe_swept_keys_total{reason="expired"}`. A finite retention must be at least `"2m"`, the ingest queue's duplicate window: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and then dropped by the queue as a copy, while the client was told it was accepted. A retention below that is refused, not raised to the minimum; so are a negative value and anything that is not a duration, such as `"30d"`, a number with no unit (`"300"` needs one: `"300s"`; `"0"` is the one exception), or a JSON number rather than a string. Hot-reloadable: a change applies to ids committed after the reload, and an id already committed keeps the expiry it was stored with. +- `dedupe.tables.
.{id_field, require_id, retention}` — per-table overrides; each entry overrides only the fields it names and inherits the rest, so a table with no `retention` keeps the tenant's (forever when the tenant sets none). A table can keep ids for a shorter time than its tenant, or for longer, or forever (`"retention": "0"`) under a finite tenant retention. ## ClickHouse diff --git a/internal/settings/settings.go b/internal/settings/settings.go index 68445543a..eeb03f358 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -143,8 +143,9 @@ type AuthConfig struct { // (dedupe.Managed, one per tenant, each a share of the one embedded Pebble // instance), so the whole block is tenant-owned. // -// id_field, require_id and retention are required here and optional per -// table: a table override inherits whichever field it doesn't name. An empty, +// id_field and require_id are required here; retention is optional, and +// missing means "0" (forever). Every field is optional per table: a table +// override inherits whichever field it doesn't name. An empty, // whitespace-only, or whitespace-padded id_field is rejected at every level, // so the effective id_field can never be empty or silently unmatchable. type DedupeConfig struct { @@ -152,9 +153,9 @@ type DedupeConfig struct { IDField *string `json:"id_field"` RequireID *bool `json:"require_id"` // Retention is how long a committed id stays a duplicate, as a Go - // duration ("720h"); "0" keeps it forever. A change applies to ids - // committed after it. - Retention *string `json:"retention"` + // duration ("720h"); "0", or leaving it out, keeps it forever. A change + // applies to ids committed after it. + Retention *string `json:"retention,omitempty"` // Tables holds per-table overrides keyed by ClickHouse table name (#222). // Names are format-checked only — existence is schema discovery's runtime // concern, same as policies.json table keys. diff --git a/internal/settings/store.go b/internal/settings/store.go index b3615efef..d707451d3 100644 --- a/internal/settings/store.go +++ b/internal/settings/store.go @@ -93,7 +93,10 @@ type Dedupe struct { func (s *Store) DedupeFor(table string) Dedupe { d := s.doc().Config.Dedupe out := Dedupe{Enabled: *d.Enabled, IDField: *d.IDField, RequireID: *d.RequireID} - retention := *d.Retention + retention := "0" + if d.Retention != nil { + retention = *d.Retention + } if td, ok := d.Tables[table]; ok { if td.IDField != nil { out.IDField = *td.IDField diff --git a/internal/settings/store_test.go b/internal/settings/store_test.go index e96898247..3487ba505 100644 --- a/internal/settings/store_test.go +++ b/internal/settings/store_test.go @@ -1,6 +1,7 @@ package settings import ( + "encoding/json" "os" "path/filepath" "testing" @@ -66,6 +67,22 @@ func TestStore_DedupeFor_Cascade(t *testing.T) { } } +// A config.json without dedupe.retention keeps ids forever, and its table +// overrides inherit that or set their own. +func TestStore_DedupeFor_RetentionMissing(t *testing.T) { + t.Parallel() + var doc map[string]map[string]any + require.NoError(t, json.Unmarshal([]byte(configJSON(`{"dedupe": {"tables": {"clicks": {"id_field": "click_id"}, "views": {"retention": "24h"}}}}`)), &doc)) + delete(doc["dedupe"], "retention") + body, err := json.Marshal(doc) + require.NoError(t, err) + s := newLoadedStore(t, map[string]string{FileConfig: string(body)}) + + assert.Equal(t, Dedupe{IDField: "event_id"}, s.DedupeFor("other"), "forever") + assert.Equal(t, Dedupe{IDField: "click_id"}, s.DedupeFor("clicks"), "inherits forever") + assert.Equal(t, Dedupe{IDField: "event_id", Retention: 24 * time.Hour}, s.DedupeFor("views")) +} + // TestStore_SeedIsValid pins that the shipped starter directory passes its // own gate: `wavehouse bootstrap` must never write something // `wavehouse validate` rejects, and the defaults are readable back. diff --git a/internal/settings/validate.go b/internal/settings/validate.go index 76352e199..52fd7637f 100644 --- a/internal/settings/validate.go +++ b/internal/settings/validate.go @@ -454,8 +454,8 @@ func (v *validator) checkIDField(path string, val *string) { // checkRetention rejects a dedupe retention that is not a duration, is // negative, or is finite but shorter than MinDedupeRetention. The short one // is refused rather than raised to the minimum, so the file never means -// something other than what it says. nil is the caller's concern, as for -// id_field. +// something other than what it says. nil is valid: forever at the tenant +// level, inherited at the table level. func (v *validator) checkRetention(path string, val *string) { if val == nil { return @@ -694,9 +694,6 @@ func (v *validator) parseConfig(data []byte) TenantConfig { if d.RequireID == nil { v.required("dedupe.require_id") } - if d.Retention == nil { - v.required("dedupe.retention") - } v.checkIDField("dedupe.id_field", d.IDField) v.checkRetention("dedupe.retention", d.Retention) // Sorted iteration keeps finding order deterministic across runs. diff --git a/internal/settings/validate_test.go b/internal/settings/validate_test.go index a6d7c5f73..04894acc6 100644 --- a/internal/settings/validate_test.go +++ b/internal/settings/validate_test.go @@ -285,7 +285,6 @@ func TestValidate_ContentRules(t *testing.T) { {"keepalive as a duration string", FileConfig, `{"stream": {"keepalive_interval": "30s"}}`, "keepalive_interval"}, {"missing dedupe.require_id", FileConfig, `{"dedupe": {"id_field": "event_id"}}`, "dedupe.require_id: required"}, {"missing dedupe.enabled", FileConfig, `{"dedupe": {"id_field": "event_id", "require_id": false, "retention": "0"}}`, "dedupe.enabled: required"}, - {"missing dedupe.retention", FileConfig, `{"dedupe": {"enabled": false, "id_field": "event_id", "require_id": false}}`, "dedupe.retention: required"}, {"dedupe.retention not a duration", FileConfig, configJSON(`{"dedupe": {"retention": "30d"}}`), `dedupe.retention: must be a duration such as "720h"`}, {"dedupe.retention a number", FileConfig, configJSON(`{"dedupe": {"retention": 3600}}`), "retention"}, {"dedupe.retention negative", FileConfig, configJSON(`{"dedupe": {"retention": "-1h"}}`), "dedupe.retention: must not be negative"}, @@ -373,6 +372,30 @@ func TestValidate_DedupeRetentionAccepted(t *testing.T) { } } +// configJSONWithout is the seed config.json less dedupe.retention. +func configJSONWithout(t *testing.T) string { + t.Helper() + var doc map[string]map[string]json.RawMessage + require.NoError(t, json.Unmarshal([]byte(configJSON(`{}`)), &doc)) + delete(doc["dedupe"], "retention") + out, err := json.Marshal(doc) + require.NoError(t, err) + return string(out) +} + +// dedupe.retention may be left out: the tenant keeps ids forever, and a +// table override may still set one. +func TestValidate_DedupeRetentionOptional(t *testing.T) { + t.Parallel() + files := validFiles() + files[FileConfig] = configJSONWithout(t) + require.NotContains(t, files[FileConfig], "retention") + doc, findings := ValidateDir(writeDir(t, files)) + require.NotNil(t, doc, "findings: %s", findingStrings(findings)) + assert.False(t, HasErrors(findings), "findings: %s", findingStrings(findings)) + assert.Nil(t, doc.Config.Dedupe.Retention) +} + // TestValidate_ClickHouseTLSPathsAreNotOpened pins that the tls block is // checked for shape only: Validate is pure and also runs on the control // plane, so paths that exist nowhere still validate, and the values reach From 76b6a9e23d8055db8e6dc9c76b34b4e0e8cc8c15 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Fri, 25 Sep 2026 12:32:49 -0400 Subject: [PATCH 04/11] docs(settings): name dedupe.retention as the one compiled default Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01EJr5tY4WQUy2sc4MbW67vL --- CHANGELOG.md | 2 +- config.yaml | 4 ++-- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/settings-directory.mdx | 2 +- internal/api/settings_test.go | 2 +- internal/app/wire.go | 3 ++- internal/settings/settings.go | 5 +++-- internal/settings/store.go | 6 +++--- 8 files changed, 14 insertions(+), 12 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index b81b90315..8806ebbb9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -79,7 +79,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Fixed -- **Dedupe claims an id, publishes, then commits it — and keys it by tenant, table and id** (`internal/dedupe/{dedupe,key,embedded,managed}.go` (+ tests), `internal/dedupe/dedupetest/` (new), `internal/api/ingest.go` (+ tests), `internal/settings/validate.go` (+ tests), `internal/testutil/mocks.go`, `internal/app/app_test.go`, `AGENTS.md`, `docs/src/content/docs/{architecture,api,deployment}.md`, `settings-directory.mdx`, `sdk/reference.md`): `CheckAndMark` is replaced by a two-phase `Reserve` → `Commit` / `Release` contract with a lease on the pending claim, and every backend now runs one conformance suite. Four bugs go with it. Two concurrent requests carrying one id no longer both publish it: Pebble's check and claim happen under one lock, and the loser answers `503` with `Retry-After` while the winner is still publishing ([#390](https://github.com/Wave-RF/WaveHouse/issues/390)). A publish that fails gives its id back, so the retry a `503` asks for is published instead of skipped as a duplicate of a record that never reached the queue ([#384](https://github.com/Wave-RF/WaveHouse/issues/384)). The same id in two tables is two ids ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s keyspace half). An explicit `null` id is a missing id — rejected under `require_id`, published un-deduped otherwise — instead of the one id `""` that made every null record after the first a duplicate ([#370](https://github.com/Wave-RF/WaveHouse/issues/370)). **Upgrade:** the key layout changes, so an id seen before the upgrade is accepted once more after it; nothing is migrated, and the old keys are left in `/pebble`, unread, deleted by the retention sweep below ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)) ([Deployment → Upgrading across the dedupe key change](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md)). The key carries the tenant's and the table's lengths rather than separators between them, so a table name is no longer refused for the bytes it holds: one holding a NUL byte dedupes in a keyspace of its own like any other. New metrics: `wavehouse_dedupe_commit_failed_total` (a published record whose id failed to commit; the claim lapses with its lease) and `wavehouse_dedupe_hashed_id_total` (an id over 1,024 bytes, stored as its SHA-256). +- **Dedupe claims an id, publishes, then commits it — and keys it by tenant, table and id** (`internal/dedupe/{dedupe,key,embedded,managed}.go` (+ tests), `internal/dedupe/dedupetest/` (new), `internal/api/ingest.go` (+ tests), `internal/settings/validate.go` (+ tests), `internal/testutil/mocks.go`, `internal/app/app_test.go`, `AGENTS.md`, `docs/src/content/docs/{architecture,api,deployment}.md`, `settings-directory.mdx`, `sdk/reference.md`): `CheckAndMark` is replaced by a two-phase `Reserve` → `Commit` / `Release` contract with a lease on the pending claim, and every backend now runs one conformance suite. Four bugs go with it. Two concurrent requests carrying one id no longer both publish it: Pebble's check and claim happen under one lock, and the loser answers `503` with `Retry-After` while the winner is still publishing ([#390](https://github.com/Wave-RF/WaveHouse/issues/390)). A publish that fails gives its id back, so the retry a `503` asks for is published instead of skipped as a duplicate of a record that never reached the queue ([#384](https://github.com/Wave-RF/WaveHouse/issues/384)). The same id in two tables is two ids ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s keyspace half). An explicit `null` id is a missing id — rejected under `require_id`, published un-deduped otherwise — instead of the one id `""` that made every null record after the first a duplicate ([#370](https://github.com/Wave-RF/WaveHouse/issues/370)). **Upgrade:** the key layout changes, so an id seen before the upgrade is accepted once more after it; nothing is migrated, and the old keys are left in `/pebble`, unread, deleted by the retention sweep (see Added) ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)) ([Deployment → Upgrading across the dedupe key change](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md)). The key carries the tenant's and the table's lengths rather than separators between them, so a table name is no longer refused for the bytes it holds: one holding a NUL byte dedupes in a keyspace of its own like any other. New metrics: `wavehouse_dedupe_commit_failed_total` (a published record whose id failed to commit; the claim lapses with its lease) and `wavehouse_dedupe_hashed_id_total` (an id over 1,024 bytes, stored as its SHA-256). - **Ingest runs in windows of 256 records over the dedupe contract, and a dedupe store that cannot answer is a `503`** (`internal/api/ingest.go` (+ tests), `internal/mq/{mq,embedded}.go` (+ tests), `internal/dedupe/key.go` (+ tests), `internal/testutil/mocks.go`, `AGENTS.md`, `docs/src/content/docs/{api,architecture,durability}.md`, `settings-directory.mdx`, `sdk/reference.md`): each window of a request is prepared, then reserved in one dedupe call, published in order, and committed in one call, so a batch costs one dedupe round trip per phase per window rather than per record — on Pebble, one commit `fsync` per window (a 1,000-record batch: four syncs instead of a thousand, 24 ms against 5.7 s of dedupe time measured with the queue stubbed). Every deduped record is published under an idempotency key (`mq.WithIdempotencyKey`, JetStream's message id, derived by `dedupe.IdempotencyKey`), and each tenant's ingest stream now keeps an explicit two-minute duplicate window, which the 30-second lease fits inside. That closes the last path of [#384](https://github.com/Wave-RF/WaveHouse/issues/384): a publish that fails with an unknown outcome (anything but a full queue) keeps its record's claim until the lease lapses instead of releasing it, and a retry after the lease but within two minutes of the first publish is dropped by the queue if the first copy was stored (a later one is stored again). A dedupe store that is not open or that reports `dedupe.ErrUnavailable` now answers `503 {"error":"dedupe store unavailable"}` with `Retry-After: 5`, which the SDK retries, rather than `500 dedupe failed`. A mid-body read error or dedupe failure now drops the open window unpublished, where records before it used to be published; `wavehouse_dedupe_commit_failed_total` and `wavehouse_ingest_dedupe_disabled_total` count records, as before, now added a window at a time. - **An insert invalidates a table's cached results under every tenant the directory holds** (`internal/app/wire.go` (+ tests), `internal/settings/registry.go` (+ tests), `AGENTS.md`, `docs/src/content/docs/{deployment,architecture,ingest-pipeline}.md`): until [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6 gives each tenant its own ClickHouse, every tenant reads the same tables, but the ingest worker — which writes every event as tenant `0`'s until story 5 — bumped only tenant `0`'s cache namespaces after an insert, so another tenant's cached query could answer stale rows for up to its TTL (an hour at most). The cache the worker invalidates through now fans each bumped namespace out to every tenant the registry knows (the new `Registry.Known`), the named one and a rejected one included — a rejected tenant comes back into service with the entries it has, so leaving it out would let a folder repaired inside a TTL serve pre-insert rows; reads are untouched, so a tenant is still never served another's cached rows. The residual, a folder removed and restored inside a TTL, is closed since #610: a tenant back on a pool after an absence has its structured-query results orphaned at once (`Cache.InvalidateTenant`). Raised by CodeRabbit on #602. - **The `?token=` strip no longer repairs a query string that does not parse** (`internal/auth/auth.go`): `bearerToken` removed a query-string token by parsing the query, deleting `token`, and re-encoding what was left — and `url.ParseQuery` skips a pair it cannot read, so the re-encoding erased that pair. A handler that parses the query strictly in order to refuse a malformed one would then see a clean query: `GET /v1/ops/pipes?tenant=acme;x=1&token=…` would have answered `200` with the default tenant's pipes. The token is read exactly as before and a query that parses is rewritten exactly as before; a query that does not parse now loses its token pairs and nothing else, byte for byte. Pinned through `api.NewRouter` with the real authenticator, since a handler-level test never runs the middleware that rewrote the URL. diff --git a/config.yaml b/config.yaml index 358124171..8675e2de6 100644 --- a/config.yaml +++ b/config.yaml @@ -67,8 +67,8 @@ auth: # query.default_max_rows / timestamp_bucket_seconds, # schema.refresh_interval, stream keepalive_interval / keepalive_buckets / # gap_window_minutes, mq.max_bytes_gb, cors.allowed_origins — and every key -# is required: the -# binary has no compiled defaults, so what's adopted is exactly what the +# is required except dedupe.retention (missing = "0", forever): the binary +# has no other compiled default, so what's adopted is exactly what the # files say. The server validates the directory at boot (invalid or missing # refuses to start) and reloads it on SIGHUP, on file change, or via # POST /v1/ops/settings/reload; a reload that fails validation keeps the diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index e9a581e88..6ab56b3d7 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -188,7 +188,7 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi - **store.go** — `Store` is a passive holder: one tenant's adopted document behind an atomic pointer, swapped by the registry, which stamps it with the id of the tenant it created the store for (`Tenant()`, how a handler names its tenant to a per-tenant resource). Consumers read typed accessors per call (`ClickHouse()`, `Auth()`, `DedupeFor(table)`, `DLQFor(table)`, `Keepalive()`, …) rather than holding values. - **registry.go** — `Registry` maps a tenant id to its `Store` and owns everything that changes one. `Open` validates and adopts at boot; `Reload` re-validates the whole directory and `ReloadTenant` one tenant's folder, serialized with each other; `AfterAdopt` hooks run after every reload the registry applied, with the tenants it adopted — none when it only rejected or removed one, which a consumer holding a resource per tenant needs to hear of too; `For(id)` and `All()` see only the tenants being served, `Known()` every tenant held, rejected ones included, and `Resolve(id)` tells a rejected tenant from an unknown one. The shape is fixed at `Open`. Flat: an invalid directory refuses boot, and a rejected reload keeps the previous snapshot. Nested: fail closed per tenant — a folder with an error finding stops being served (the store keeps its document for requests already admitted, and gets the next good one) while the rest carry on; a whole-directory reload mirrors the folders, down to none (an emptied directory is not a change of shape); and a finding about the directory itself refuses boot or rejects the reload whole, leaving every tenant as it was. The tenant map is replaced whole by a reload, so a lookup is one lock-free load. - **watch.go** — `Registry.Watch`, which `internal/app` starts for a flat directory only: fsnotify on the *directory* (not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost), debounced into one reload; reloads once as soon as the watch exists so an edit between the boot read and the watch is never missed. `SIGHUP` and the reload endpoint funnel through the same serialized `Reload`. -- **seed.go** / **seed/** — The embedded (`go:embed`) starter directory with every key at its default. The binary carries no compiled defaults: `wavehouse bootstrap [dir]` writes this seed, and the compose stack and e2e fixture ship copies of it. +- **seed.go** / **seed/** — The embedded (`go:embed`) starter directory with every key at its default. The binary carries no compiled defaults except that a missing `dedupe.retention` means `"0"`: `wavehouse bootstrap [dir]` writes this seed, and the compose stack and e2e fixture ship copies of it. ### `tenant/` — Tenant Identifier diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 55a8e0d9c..9cb52b4d3 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -11,7 +11,7 @@ Boot config — the YAML file and `WH_*` environment variables on the [Configura The settings directory holds WaveHouse's file-based settings as exactly four JSON documents: [`roles.json`](#rolesjson), [`policies.json`](#policiesjson), [`pipes.json`](#pipesjson), and [`config.json`](#configjson-keys). The files are the only write path — standalone, you edit them on the host; on WaveHouse Cloud the control plane writes them — and there is no API that writes back to them. Every file must exist (an empty document is `{}` — a missing file always means deletion or a wrong path, never "defaults"), and any other entry in the directory is an error, so a typoed filename or a stray backup fails loudly instead of being silently ignored. Dot-prefixed entries are the one exception: editor swap files and the `..data` machinery Kubernetes ConfigMap mounts publish through are ignored. -Create one with `wavehouse bootstrap [dir]`: it writes all four files with every key at its default and refuses a non-empty directory, so an existing settings directory is never overwritten. The binary carries no compiled defaults — the seed is the one place they live, and what the server adopts is exactly what the files say. The seed ships no policy (`policies.json` is `{}`, `roles.json` and `pipes.json` are empty lists), so a freshly bootstrapped directory boots fail-closed — every request is denied until you write a policy. The container images ship no settings directory: they preset `WH_SETTINGS_DIR=/app/settings` and expect a bind mount there — a host directory you wrote with `bootstrap` (the reference compose file mounts the checked-in `deployments/compose/settings/`, the seed with `clickhouse.addr` pointed at the `clickhouse` service and a permissive `public` trial policy in `policies.json` / `roles.json`). A bind mount, not a named volume: the images are distroless, with no shell to edit files inside a volume. A missing mount refuses to boot rather than running on defaults nobody chose. +Create one with `wavehouse bootstrap [dir]`: it writes all four files with every key at its default and refuses a non-empty directory, so an existing settings directory is never overwritten. The binary carries no compiled defaults but one — a missing `dedupe.retention` means `"0"`, forever — so the seed is where the defaults live, and what the server adopts is exactly what the files say. The seed ships no policy (`policies.json` is `{}`, `roles.json` and `pipes.json` are empty lists), so a freshly bootstrapped directory boots fail-closed — every request is denied until you write a policy. The container images ship no settings directory: they preset `WH_SETTINGS_DIR=/app/settings` and expect a bind mount there — a host directory you wrote with `bootstrap` (the reference compose file mounts the checked-in `deployments/compose/settings/`, the seed with `clickhouse.addr` pointed at the `clickhouse` service and a permissive `public` trial policy in `policies.json` / `roles.json`). A bind mount, not a named volume: the images are distroless, with no shell to edit files inside a volume. A missing mount refuses to boot rather than running on defaults nobody chose. Check a directory with `wavehouse validate [dir]`. Both commands resolve the directory the same way — the argument, falling back to `WH_SETTINGS_DIR`, and a usage error (exit `2`) with neither — so the path you seed is the path you validate, and inside the container images (which preset `WH_SETTINGS_DIR=/app/settings`) both work with no argument at all. `validate` validates without starting the server (JSON syntax including unknown fields and duplicate keys, per-file shape rules including the required keys, and cross-file role references), prints every finding in one pass, and exits `0` for valid (warnings allowed), `1` for invalid, `2` for usage — so operators and CI can gate a settings change before it reaches a running instance. diff --git a/internal/api/settings_test.go b/internal/api/settings_test.go index 24fa80f12..462cbc8e6 100644 --- a/internal/api/settings_test.go +++ b/internal/api/settings_test.go @@ -16,7 +16,7 @@ import ( "github.com/stretchr/testify/require" ) -// fullConfig is a complete config.json (every key is required) with the +// fullConfig is a complete config.json (every key set) with the // given query.default_max_rows. func fullConfig(maxRows int) string { return fmt.Sprintf(`{"clickhouse": {"addr": "localhost:9000", "http_port": 8123, "http_scheme": "http", "database": "default", "username": "default", "query_timeout": 30, "tls": {"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": {}, "max_open_conns": 10, "max_idle_conns": 5}, "auth": {"jwks_url": "", "role_claim": "role"}, "dedupe": {"enabled": false, "id_field": "event_id", "require_id": false, "retention": "0"}, "dlq": {"enabled": true}, "query": {"default_max_rows": %d, "timestamp_bucket_seconds": 60}, "schema": {"refresh_interval": 60}, "stream": {"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": 15}, "mq": {"max_bytes_gb": 1}, "cors": {"allowed_origins": ["*"]}}`, maxRows) diff --git a/internal/app/wire.go b/internal/app/wire.go index ac494bea7..2678646a1 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -52,7 +52,8 @@ func withoutContext(release func() error) func(context.Context) error { // configuration (dedupe, dlq, query, schema, stream, cors — see // settings.TenantConfig). Required: config.Validate already rejected an // empty settings.dir, and an invalid directory refuses boot. The binary -// carries no compiled defaults; `wavehouse bootstrap` writes the seed. A +// carries no compiled defaults but a missing dedupe.retention ("0"); +// `wavehouse bootstrap` writes the seed. A // *reload* of an invalid directory merely keeps the previous snapshot. A // nested directory (one folder per tenant, #583) fails closed per tenant // instead, at boot and on reload alike: see settings.Registry. diff --git a/internal/settings/settings.go b/internal/settings/settings.go index eeb03f358..4d5fa7a64 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -65,8 +65,9 @@ type PipesFile struct { // `cache.l1_max_cost`), listeners, the observability // exporters — and the secrets (`clickhouse.password`, `auth.jwt_secret`, // `auth.operator_key`), which never belong in a tracked JSON file. Every -// block and every top-level key inside it is REQUIRED: the binary carries no -// compiled defaults, so the adopted snapshot is exactly what the files say. +// block and every top-level key inside it is REQUIRED, but dedupe.retention +// (missing means "0", forever): the binary carries no other compiled default, +// so the adopted snapshot is exactly what the files say. // Defaults live in the seed directory (see Seed) that `wavehouse // bootstrap` writes. The fields are pointers only so Validate can tell // "absent" from the zero value and report it by path. diff --git a/internal/settings/store.go b/internal/settings/store.go index d707451d3..c2609276f 100644 --- a/internal/settings/store.go +++ b/internal/settings/store.go @@ -16,9 +16,9 @@ import ( // accessors below each resolve from a single snapshot load, so a reload lands // between lookups, never inside one. // -// There are no compiled defaults here on purpose: every key is required by -// Validate, so the snapshot is exactly what the files said when they were -// adopted. Defaults live in the seed directory (Seed / WriteSeed). +// There are no compiled defaults here on purpose, but one: every key but +// dedupe.retention (missing means "0", forever) is required by Validate, so +// the snapshot is exactly what the files said when they were adopted. Defaults live in the seed directory (Seed / WriteSeed). type Store struct { // tenant is the id the Registry created the store for; the zero value // only for a Store built outside a Registry (tests). From bea9e70150400f08f30d4b6bd29dd78f92dc3fce Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Fri, 25 Sep 2026 12:34:42 -0400 Subject: [PATCH 05/11] docs(settings): the last every-key-required comment; rewrap Store's Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01EJr5tY4WQUy2sc4MbW67vL --- internal/settings/store.go | 3 ++- internal/settings/validate_test.go | 5 +++-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/internal/settings/store.go b/internal/settings/store.go index c2609276f..a03546411 100644 --- a/internal/settings/store.go +++ b/internal/settings/store.go @@ -18,7 +18,8 @@ import ( // // There are no compiled defaults here on purpose, but one: every key but // dedupe.retention (missing means "0", forever) is required by Validate, so -// the snapshot is exactly what the files said when they were adopted. Defaults live in the seed directory (Seed / WriteSeed). +// the snapshot is exactly what the files said when they were adopted. +// Defaults live in the seed directory (Seed / WriteSeed). type Store struct { // tenant is the id the Registry created the store for; the zero value // only for a Store built outside a Registry (tests). diff --git a/internal/settings/validate_test.go b/internal/settings/validate_test.go index 7af9214d3..96e1ff60a 100644 --- a/internal/settings/validate_test.go +++ b/internal/settings/validate_test.go @@ -33,8 +33,9 @@ func validFiles() map[string]string { // configJSON returns the seed config.json with patch merged over it, one // level deep (a patched block's keys replace the seed's, the rest of the -// block is kept). Every key is required, so tests that care about one key -// build a complete document from the seed rather than repeating all of them. +// block is kept). Every key but dedupe.retention is required, so tests that +// care about one key build a complete document from the seed rather than +// repeating all of them. func configJSON(patch string) string { seed, err := Seed() if err != nil { From 5d77211de17e38810cbbf3335a5fd99012fedbaa Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Fri, 25 Sep 2026 13:37:28 -0400 Subject: [PATCH 06/11] test(dedupe): a legacy value under a current key reads as absent The version-0 read test planted only a tenant-NUL-id key, which the text layout never looks up, so it passed whatever Reserve made of a legacy value. It now also plants a bare legacy id that spells a current key and checks it reads as absent and is overwritten by the commit. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01EJr5tY4WQUy2sc4MbW67vL --- docs/src/content/docs/deployment.md | 2 +- internal/dedupe/embedded_test.go | 20 ++++++++++++++------ internal/dedupe/sweep.go | 2 +- 3 files changed, 16 insertions(+), 8 deletions(-) diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index ca23f0fe6..6c103c0df 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -419,7 +419,7 @@ WaveHouse discovers this schema on startup and refreshes it every `schema.refres ## Upgrading across the dedupe key change -The dedupe key now carries the table as well as the tenant ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)), so **an id deduped before the upgrade is not recognized after it**: a record carrying it is accepted once more. Nothing is migrated. The old keys are never read, and the dedupe sweep deletes them: its first pass runs about a minute after the instance opens, and `wavehouse_dedupe_swept_keys_total{reason="version_0"}` counts them ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)). Pebble returns their disk space as it compacts, not at once. Only a tenant with `dedupe.enabled` on is affected, and only by a record sent both before and after the upgrade — typically a producer retrying across the restart. To avoid duplicate rows, let retrying producers finish, or pause them, before upgrading. +The dedupe key now carries the table as well as the tenant ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)), so **an id deduped before the upgrade is not recognized after it**: a record carrying it is accepted once more. Nothing is migrated. The old keys never count as seen, and the dedupe sweep deletes them: its first pass runs about a minute after the instance opens, and `wavehouse_dedupe_swept_keys_total{reason="version_0"}` counts them ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)). Pebble returns their disk space as it compacts, not at once. Only a tenant with `dedupe.enabled` on is affected, and only by a record sent both before and after the upgrade — typically a producer retrying across the restart. To avoid duplicate rows, let retrying producers finish, or pause them, before upgrading. The same release adds an optional **`dedupe.retention`** key. No upgrade step is needed: a `config.json` without it keeps every id forever, as before. See [Deduplication](/settings-directory#deduplication) for a finite one. diff --git a/internal/dedupe/embedded_test.go b/internal/dedupe/embedded_test.go index 5015318f4..8dfd29733 100644 --- a/internal/dedupe/embedded_test.go +++ b/internal/dedupe/embedded_test.go @@ -140,17 +140,25 @@ func TestEmbedded_OpenFailure(t *testing.T) { assert.True(t, e.Open()) } -// Keys from before the table joined the key (#222) are never read: an id -// seen then is accepted once more after the upgrade, the documented cost of -// the new layout. -func TestEmbedded_VersionZeroKeysAreNotRead(t *testing.T) { +// Keys from before the table joined the key (#222) never count: an id seen +// then is accepted once more after the upgrade, the documented cost of the +// new layout. A tenant ‖ NUL ‖ id key is never looked up; a bare v0.1.0 id +// that spells a current key is, and its 8-byte value reads as absent. +func TestEmbedded_VersionZeroKeysDoNotCount(t *testing.T) { t.Parallel() e := NewEmbedded(t.TempDir()) m := switchedOn(t, e, "acme") require.NoError(t, e.db.Set([]byte("acme\x00e1"), make([]byte, 8), pebble.Sync)) - dup, err := mark(context.Background(), m, "e1") + stale := AppendKey(nil, KeyPrefix("acme"), Key{Table: "events", ID: "e2"}) + require.NoError(t, e.db.Set(stale, make([]byte, 8), pebble.Sync)) + for _, id := range []string{"e1", "e2"} { + dup, err := mark(context.Background(), m, id) + require.NoError(t, err) + assert.False(t, dup, id) + } + dup, err := mark(context.Background(), m, "e2") require.NoError(t, err) - assert.False(t, dup) + assert.True(t, dup, "the commit overwrote the stale value") } // A claim nobody commits, releases or reserves again leaves memory at the diff --git a/internal/dedupe/sweep.go b/internal/dedupe/sweep.go index 444ec178d..700186847 100644 --- a/internal/dedupe/sweep.go +++ b/internal/dedupe/sweep.go @@ -75,7 +75,7 @@ func (e *Embedded) startSweep(db *pebble.DB) (stop func()) { } // sweep makes one pass over the whole instance, deleting keys whose -// retention has ended and version-0 keys, which nothing reads: those from +// retention has ended and version-0 keys, which never count: those from // before ids were keyed by table (tenant ‖ 0x00 ‖ id, or the bare id before // that). They are told apart by value, since a bare id may be any bytes, a // current key's included: only commits are stored, and every commit has the From e52b336c3c7b41d3d66a608102093bda30f4c330 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Sat, 26 Sep 2026 04:53:59 -0400 Subject: [PATCH 07/11] fix(dedupe): sweep holds the commit lock only to re-check and delete A sweep chunk held commitMu while it iterated to its 1,024th live key. Pebble skips point tombstones inside Next, so a chunk that started over a run of them (a tenant's version-0 block deleted by the first pass, or a table whose ids had all expired) held every Commit, and with it every deduped ingest response, for the whole run, on every pass until the run was compacted. The chunk now reads without the lock and collects the keys it would delete, then takes commitMu only to re-read each one and delete those still expired or version-0, without fsync as before. A Commit waits for at most 1,024 point reads and one unsynced batch, however many tombstones lie between the keys. The mid-chunk test splits in two, one per gap a Commit can land in: after the unlocked read, where the re-read keeps the key, and after the re-read, where the lock holds the Commit off until the delete is done. Each fails when its guard is removed. A new test races Commits against a chunk that starts over 300,000 tombstones: under the race detector the old chunk kept one waiting 241-278 ms, the new one at most 11 ms. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01G6Cz4H5k1spJAPk5CZeGds --- CHANGELOG.md | 2 +- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/durability.md | 2 +- internal/dedupe/embedded.go | 13 +-- internal/dedupe/sweep.go | 119 ++++++++++++++++++-------- internal/dedupe/sweep_test.go | 87 ++++++++++++++++--- 6 files changed, 172 insertions(+), 53 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7b7935f03..9c9feadbf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,7 +29,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Settings-directory validation — `wavehouse validate [dir]`** (`internal/settings/` (new: `settings.go`, `validate.go`, `decode.go`, `finding.go`, + tests), `cmd/wavehouse/validate.go` (new, + tests), `cmd/wavehouse/main.go`): first piece of the file-based control plane (settings live in a directory of JSON documents — `roles.json`, `policies.json`, `pipes.json`, `config.json` — that a running instance will hot-reload; this change is validation-only — boot loading and reload wiring land separately). `settings.Validate(dir)` is the single gate every consumer of the directory runs: deliberately pure (no network, no ClickHouse — table/column existence stays with schema discovery, per Bring-Your-Own-Schema), and it collects **all** findings in one pass instead of failing on the first. Checks, layered: the directory holds exactly the four files (a missing file is an error — an empty document is `{}`, so absence always means deletion or a wrong path; any unexpected entry — file or directory — is an error so a typoed `polices.json` or a stray backup can't be silently ignored; dot-prefixed entries are the one carve-out, since erroring on vim swap files or the `..data` machinery Kubernetes ConfigMap mounts publish through would break hand editing and the cloud fan-out's mount pattern alike); strict JSON syntax (unknown fields rejected — the JSON form of the retired-config-key trap; empty/truncated files rejected, never read as an empty document; a leading UTF-8 byte order mark named as such instead of surfacing as a cryptic invalid-character error; a directory, unreadable file, or non-regular file (a FIFO would hang the read forever waiting for a writer; a stat gate rejects it — following symlinks, so Kubernetes ConfigMap mounts' symlink layout still passes) squatting on a settings filename named as the one real problem, not double-reported as "missing"; a top-level `null` rejected — the one well-formed document that decodes into a zero value without error, so it would silently read as "no settings"; trailing content rejected; duplicated object keys detected by a token-level pass, since `encoding/json` silently keeps the last copy); per-file shape rules (role names non-empty/unique, pipe names/SQL/param types, `config.json` bounds mirroring boot-config validation — its sections are the *tenant-owned* behavioral tunables (dedupe id_field/require_id plus per-table overrides under `dedupe.tables` — each entry overrides only the fields it names, resolving table → global → compiled default per field, so the effective id_field can never be empty — an explicit empty, whitespace-only, or whitespace-padded id_field is rejected at both levels, since an exact-match JSON key lookup would silently miss every row ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s shape, unblocked by the file design since table names are runtime-resolved like policy grants); query default_max_rows, schema refresh_interval, CORS origins); platform-owned knobs like the SSE keepalives deliberately stay boot config); and cross-file referential integrity (every role a policy grant, `default_role`/`admin_role`, or pipe allowlist references must be declared in `roles.json`; an empty role string in a grant or allowlist is named as such — it matches no request and authorizes nobody). Warnings don't invalidate: a grant scoping the admin role (an unconditional bypass — dead config), `default_role` = admin, and a `default` on a required pipe parameter are flagged but legal. An empty `policies.json` means no policy — fail closed, matching deleted-policy semantics — and draws a warning naming the total lockout, so it announces itself at validation time instead of one 403 at a time. The CLI (`cmd/wavehouse/validate.go`, following the `health` subcommand pattern) takes the directory as an argument or from `WH_SETTINGS_DIR`, prints findings, and exits 0/1/2 (valid/invalid/usage) so CI and operators can gate config changes before they reach a running instance. The dispatch in `main.go` also grows `help` and `version` subcommands, and an unknown command is now a usage error instead of silently falling through and starting the server (`wavehouse validat` booting a listener is not a typo anyone wants); each subcommand parses its arguments with a stdlib `flag.FlagSet`, so `wavehouse -h` prints command-specific help and a stray flag or argument is a usage error rather than being silently swallowed. `WH_SETTINGS_DIR` has a single authority: `config.EnvSettingsDir`, with a reflection test pinning the `settings.dir` struct tag to it. The directory's location joins boot config as `settings.dir` (`WH_SETTINGS_DIR`; `internal/config/config.go`, `config.yaml`, `docs/src/content/docs/configuration.mdx`) — boot-tier by necessity, since it's the pointer the reload machinery follows; no default, same silent-misconfiguration reasoning as `policy.file_path`. - **Docs-site analytics for search, code copies, 404s, docs section, and live-demo connectivity** (`docs/src/components/DocsTracking.astro` (new), `docs/src/components/{PostHog,Footer,LiveDemo}.astro`): the site tracked its own CTAs but nothing a reader did on the way to one, so the questions that decide what to write next — what people search for and *don't* find, which snippets get copied, which dead links keep getting followed — had no data behind them. `docs_search` fires a second after the query settles rather than once per keystroke, carrying `query` and `result_count` read off Pagefind's own results message (the rendered list is capped at its page size, so counting the DOM would under-report); `result_count: 0` is the event worth having. `code_copied` (`page`, `language`) watches Expressive Code's copy buttons from the document rather than re-binding every code block on every navigation — the hero's install chip is not an EC block and keeps its own `hero_install_copied`. `docs_404` (`path`, `referrer`) turns broken inbound links into a list instead of a hunch. A `doc_section` property (the first path segment, `home` for `/`) puts every event in a docs area without each tracker carrying its own copy; it's stamped at capture time by a `before_send` hook in `posthog.init()` rather than `register()`, because a queued `register()` replays only after init has already captured the first hard-load `$pageview` — which would then carry the previous visit's persisted value — and `history_change` navigations update the URL before capture fires, so reading `location` in the hook is always current. `live_demo_connected` fires once per mount when the hero's SSE feed comes up rather than on its first row — named for what it measures (the demo backend answered), since a quiet minute on the repo is not a disengaged reader. The three site-wide trackers share one new `DocsTracking.astro` rendered from the footer (like `MermaidZoom` / `ScrollHints`) and delegate from `document`, since Pagefind, Expressive Code, and the 404 route all own their own markup — some of it created after page load. -- **Dedupe retention per tenant and table, and a sweep that deletes expired ids** (`internal/settings/{settings,validate,store}.go` (+ tests), `internal/settings/seed/config.json`, `internal/dedupe/{embedded,sweep}.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/testutil/mocks.go`, `deployments/compose/settings/config.json`, `tests/e2e/fixtures/settings/config.json`, `config.yaml`, `docs/src/content/docs/{settings-directory.mdx,deployment,durability,architecture}.md`, `AGENTS.md`): [#220](https://github.com/Wave-RF/WaveHouse/issues/220). `config.json` gains an optional **`dedupe.retention`** key, overridable per table in `dedupe.tables.
.retention`: how long a committed id stays a duplicate, as a duration string (`"720h"`), or `"0"` to keep it forever, which is the seed value and the behaviour before this release. A `config.json` without the key keeps ids forever, and a table override without one inherits the tenant's, so an existing directory needs no change. A finite retention below `"2m"`, the ingest queue's duplicate window, is refused rather than raised to the minimum: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and dropped by the queue as a copy while the client was told it was accepted. It is hot-reloadable and read per record like `id_field`; a change applies to ids committed after it, and a reload landing mid-window commits each record with the retention it was prepared under. The embedded Pebble store already treated an expired id as new; it now also deletes expired keys, and the version-0 keys the key-layout change left behind, in a background sweep that starts a minute after the instance opens and repeats hourly. It reads 1,024 keys per chunk and deletes the expired and version-0 ones, without fsync, under a lock `Commit` also takes, so an id committed again after the sweep read it is never deleted. New metric `wavehouse_dedupe_swept_keys_total{reason="expired"|"version_0"}`. `settings.Store.DedupeFor` now returns a `settings.Dedupe` struct rather than three values. +- **Dedupe retention per tenant and table, and a sweep that deletes expired ids** (`internal/settings/{settings,validate,store}.go` (+ tests), `internal/settings/seed/config.json`, `internal/dedupe/{embedded,sweep}.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/testutil/mocks.go`, `deployments/compose/settings/config.json`, `tests/e2e/fixtures/settings/config.json`, `config.yaml`, `docs/src/content/docs/{settings-directory.mdx,deployment,durability,architecture}.md`, `AGENTS.md`): [#220](https://github.com/Wave-RF/WaveHouse/issues/220). `config.json` gains an optional **`dedupe.retention`** key, overridable per table in `dedupe.tables.
.retention`: how long a committed id stays a duplicate, as a duration string (`"720h"`), or `"0"` to keep it forever, which is the seed value and the behaviour before this release. A `config.json` without the key keeps ids forever, and a table override without one inherits the tenant's, so an existing directory needs no change. A finite retention below `"2m"`, the ingest queue's duplicate window, is refused rather than raised to the minimum: every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and dropped by the queue as a copy while the client was told it was accepted. It is hot-reloadable and read per record like `id_field`; a change applies to ids committed after it, and a reload landing mid-window commits each record with the retention it was prepared under. The embedded Pebble store already treated an expired id as new; it now also deletes expired keys, and the version-0 keys the key-layout change left behind, in a background sweep that starts a minute after the instance opens and repeats hourly. It reads 1,024 keys per chunk without a lock, then re-reads the expired and version-0 ones under a lock `Commit` also takes and deletes, without fsync, those that still are, so an id committed again after the sweep read it is never deleted, and a `Commit` waits for at most one chunk's re-reads, never for the deleted keys a chunk steps over. New metric `wavehouse_dedupe_swept_keys_total{reason="expired"|"version_0"}`. `settings.Store.DedupeFor` now returns a `settings.Dedupe` struct rather than three values. ### Changed diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 642d25a03..b7fa3db58 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -125,7 +125,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ - **dedupe.go** — the `Deduplicator` contract, two-phase: `Reserve(ctx, keys, lease)` answers one `Claim` per `Key{Table, ID}`, in order — `Claimed` (first sighting: the caller now holds a pending claim), `Duplicate` (committed earlier, or repeated earlier in the same call) or `InFlight` (another request holds a live claim) — and is atomic per key across every process sharing the backend; `Commit(ctx, claims, retention)` makes the published ids duplicates (retention `0` = forever); `Release(ctx, claims)` gives back ids whose records were definitely not published (a refused or never-sent publish; one whose outcome is unknown is left to lapse instead). A claim neither committed nor released lapses after its lease, so a request that dies mid-publish never strands an id. There is deliberately no read-only check: a separate read is how [#390](https://github.com/Wave-RF/WaveHouse/issues/390) happened. - **key.go** — the key every backend stores, as text: `/
/` (for example `acme/clicks/evt%2D123`, [#222](https://github.com/Wave-RF/WaveHouse/issues/222)). The table and id are escaped by `internal/keyenc` — the escaping NATS subject tokens use — which never writes `/`, and a tenant id cannot hold one, so a table name may hold any byte, NUL included, and neither tenants nor tables share ids; a key is ASCII, so it reads as-is in a console and is a valid DynamoDB String. An id whose escaped form is over 1,024 bytes is stored as `#` plus its SHA-256 in hex (`#` is never written by the escaping), counted by `wavehouse_dedupe_hashed_id_total`. -- **embedded.go** — `Embedded`, the [Pebble](https://github.com/cockroachdb/pebble) (embedded key-value store) implementation: every tenant's seen ids in one instance at `data_dir/pebble` ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 3). Pending claims live in memory beside it, in 64 locked shards: one process owns the instance, so a crash forgetting them is every lease lapsing at once, and the shard lock makes check-and-claim atomic. `Commit` writes every claim it is given in one batch and one fsync, each value carrying its expiry (`0` = never), which `Reserve` honors on read. A background sweep (`sweep.go`), started when the instance opens and stopped before it closes, deletes expired keys and the version-0 keys from before the table joined the key — told apart by their value, which is never a current commit's, since a bare id from before tenants led the key could spell a current one ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)): a minute after opening, then hourly, 1,024 keys per chunk, holding a lock `Commit` also takes, so a key re-committed after the sweep read it is never deleted; `wavehouse_dedupe_swept_keys_total{reason}` counts what it deletes. `NewEmbedded(dataDir)` opens nothing; `Tenant(id)` is the `Factory` a `Stores` takes, building the tenant's `Managed` over its share of the instance, which opens with the first tenant store switched on and closes with the last one switched off. `Stats` reports the instance's figures for the system gauges, nil while it is closed. Pebble is per process: two pods on it do not share seen ids. +- **embedded.go** — `Embedded`, the [Pebble](https://github.com/cockroachdb/pebble) (embedded key-value store) implementation: every tenant's seen ids in one instance at `data_dir/pebble` ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 3). Pending claims live in memory beside it, in 64 locked shards: one process owns the instance, so a crash forgetting them is every lease lapsing at once, and the shard lock makes check-and-claim atomic. `Commit` writes every claim it is given in one batch and one fsync, each value carrying its expiry (`0` = never), which `Reserve` honors on read. A background sweep (`sweep.go`), started when the instance opens and stopped before it closes, deletes expired keys and the version-0 keys from before the table joined the key — told apart by their value, which is never a current commit's, since a bare id from before tenants led the key could spell a current one ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)): a minute after opening, then hourly, 1,024 keys per chunk, read without a lock, so the deleted keys a chunk steps over (Pebble keeps them until it compacts) never hold up a `Commit`, then re-read under a lock `Commit` also takes and deleted only if still expired or version-0, so a key re-committed after the sweep read it is never deleted; `wavehouse_dedupe_swept_keys_total{reason}` counts what it deletes. `NewEmbedded(dataDir)` opens nothing; `Tenant(id)` is the `Factory` a `Stores` takes, building the tenant's `Managed` over its share of the instance, which opens with the first tenant store switched on and closes with the last one switched off. `Stats` reports the instance's figures for the system gauges, nil while it is closed. Pebble is per process: two pods on it do not share seen ids. - **managed.go** — `Managed` wraps one store — opened through the function `NewManaged` takes, so the switch semantics are the same for every backend — behind the hot-reloadable `dedupe.enabled` switch: `Apply(enabled)` opens or closes it, idempotently, and in-flight calls are serialized against the swap, so flipping the key is a reload, not a restart. Every call returns `ErrDisabled` while switched off (the ingest handler publishes un-deduped and counts it — a reload-window race, not a mode) and `ErrUnavailable` while switched on but not open (ingest fails closed). `Reserve` also reads a lease `<= 0` as `DefaultLease` and collapses a key repeated in one call before the backend sees it, once for every backend — so a backend may assume distinct keys, a positive lease, and only `Claimed` claims in `Commit` and `Release`. - **dedupetest/** — the conformance suite every backend runs: `Run(t, newHarness)` drives the contract above through a backend's `Factory` (claim, commit, release, lease lapse, retention, one claim among concurrent reserves from two clients, keyspaces, input order, late commit, stale release, failure mid-call); `Harness` optionally injects a clock and a mid-call failure. `Mark` is the old check-and-mark in one call, for tests that only need an id seen. - **stores.go** — `Stores` is one `Managed` per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7), built on first use through a `Factory` (`func(tenant.ID) *Managed`) — whether tenants share a backend is the factory's business (`Embedded.Tenant` puts them all in one Pebble instance), with nothing that holds the `Stores` changing. `For(id)` returns a tenant's store, built closed so a tenant adopted a moment ago answers `ErrDisabled` rather than having no store; `Retain(keep)` closes and forgets the stores of tenants no longer served, touching nothing on disk; `Close()` closes every store. `internal/app` drives it from the registry's `AfterAdopt` hook. diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index 32e9f5b2e..48fc88173 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -62,7 +62,7 @@ The tell for a commit-cadence problem (ZFS-without-SLOG, noisy-neighbor VM host) With [deduplication](/settings-directory#deduplication) on, a `200` also means the records' ids were committed to the dedupe store, or, if that commit failed, that the failure was counted by `wavehouse_dedupe_commit_failed_total` and the ids lapse with their lease. On the embedded Pebble store that commit is an `fsync` of its own. It is taken once per window of up to 256 records of a request, after the window's publishes, rather than once per record: a 1,000-record batch costs four dedupe syncs, not a thousand. Measured with `BenchmarkIngest_DedupBatchOnPebble` on a developer laptop, with the queue stubbed out so only the dedupe store touched disk, the dedupe work for that batch took 24 ms windowed against 5.7 s one record at a time; the JetStream publishes' own fsyncs come on top. A single-record request still pays one sync for its publish and one for its commit. -With a finite `dedupe.retention`, expired ids are deleted by a background sweep, an hour apart. Its deletes are not fsynced (a delete lost to a crash is redone by the next pass), so it adds no sync to the ingest path; it reads 1,024 keys at a time, deleting the expired ones, and a commit that arrives mid-chunk waits for that chunk. An expired id is already treated as new by the next claim of it, sweep or no sweep, so retention never depends on the sweep having run. +With a finite `dedupe.retention`, expired ids are deleted by a background sweep, an hour apart. Its deletes are not fsynced (a delete lost to a crash is redone by the next pass), so it adds no sync to the ingest path. It reads 1,024 keys at a time without holding up commits, then re-reads the expired ones and deletes those still expired; a commit waits only for that last step, at most 1,024 point reads and one unsynced write, however many deleted keys the read stepped over. An expired id is already treated as new by the next claim of it, sweep or no sweep, so retention never depends on the sweep having run. A publish can also fail after JetStream stored the event (a timeout on the ack). The record's id is then left to lapse with its 30-second dedupe lease rather than given back, and every deduped record is published under an idempotency key derived from its tenant, table and id, which each tenant's ingest stream remembers for two minutes after the first publish. A retry after the lease but inside those two minutes is therefore dropped by the stream rather than stored twice; one later than that is stored again. The 30-second lease sits well inside those two minutes, so a prompt retry is covered. For the same reason a finite `dedupe.retention` must be at least those two minutes: an id re-sent after a shorter retention ended would be claimed again, then dropped by the stream as a copy while the client was told it was accepted. Settings validation refuses one below it. diff --git a/internal/dedupe/embedded.go b/internal/dedupe/embedded.go index d0fe87ac9..5eece3f3b 100644 --- a/internal/dedupe/embedded.go +++ b/internal/dedupe/embedded.go @@ -35,8 +35,9 @@ type Embedded struct { open int // tenant stores open over db stopSweep func() // stops db's sweep - // commitMu is read-held by Commit and held by a sweep chunk, so a sweep - // never deletes a key a Commit rewrote after the sweep read it. + // commitMu is read-held by Commit and held by a sweep chunk while it + // re-reads and deletes, so a sweep never deletes a key a Commit rewrote + // after the sweep read it. commitMu sync.RWMutex sweepFirst time.Duration sweepEvery time.Duration @@ -47,9 +48,11 @@ type Embedded struct { // readHook, when set, runs before each Pebble read in Reserve; a test // makes it fail to exercise Reserve's all-or-nothing error path. readHook func() error - // sweepHook, when set, runs in a sweep chunk between reading its keys and - // deleting them; a test races a Commit into that gap. - sweepHook func() + // sweepScanHook and sweepDeleteHook, when set, run in a sweep chunk: + // between its unlocked read and its re-read, and between its re-read and + // its delete. A test races a Commit into each gap. + sweepScanHook func() + sweepDeleteHook func() } // NewEmbedded returns the embedded implementation under dataDir. Nothing is diff --git a/internal/dedupe/sweep.go b/internal/dedupe/sweep.go index 700186847..f1c4ca180 100644 --- a/internal/dedupe/sweep.go +++ b/internal/dedupe/sweep.go @@ -3,6 +3,7 @@ package dedupe import ( "bytes" "context" + "errors" "fmt" "log/slog" "time" @@ -20,9 +21,9 @@ import ( const ( sweepInterval = time.Hour sweepFirstDelay = time.Minute - // sweepChunk keys are read and deleted per lock hold, with sweepPause - // between chunks: at most ~100k keys a second, and a Commit never waits - // longer than one chunk. + // sweepChunk keys are read per chunk, with sweepPause between chunks: at + // most ~100k keys a second. A Commit waits only for a chunk's re-reads + // and deletes, never for its read, however many tombstones it skips. sweepChunk = 1024 sweepPause = 10 * time.Millisecond ) @@ -98,21 +99,41 @@ func (e *Embedded) sweep(ctx context.Context, db *pebble.DB) (sweepResult, error } // sweepChunk deletes the sweepable keys among the next sweepChunk keys from -// from, returning where the next chunk starts (nil at the end). It holds -// commitMu, so no Commit lands between reading a key and deleting it: a key -// re-committed after it expired is never deleted with its new value. +// from, returning where the next chunk starts (nil at the end). It reads them +// without commitMu, since Pebble skips the tombstones between keys inside the +// read and a run of them left by an earlier pass would otherwise hold every +// Commit for its whole length. func (e *Embedded) sweepChunk(ctx context.Context, db *pebble.DB, from []byte, res *sweepResult) ([]byte, error) { - e.commitMu.Lock() - defer e.commitMu.Unlock() - now := e.now() + candidates, next, err := sweepCandidates(db, from, e.now()) + if err != nil || len(candidates) == 0 { + return next, err + } + if e.sweepScanHook != nil { + e.sweepScanHook() + } + expired, v0, err := e.deleteSweepable(db, candidates) + if err != nil { + return nil, err + } + res.Expired += int(expired) + res.Version0 += int(v0) + if expired > 0 { + sweptKeysCounter.Add(ctx, expired, metric.WithAttributes(attribute.String(sweptAttribute, sweptExpired))) + } + if v0 > 0 { + sweptKeysCounter.Add(ctx, v0, metric.WithAttributes(attribute.String(sweptAttribute, sweptVersion0))) + } + return next, nil +} + +// sweepCandidates reads the next sweepChunk keys, starting at from, and +// returns those sweepable at now and where the next chunk starts (nil at the +// end). +func sweepCandidates(db *pebble.DB, from []byte, now time.Time) (candidates [][]byte, next []byte, err error) { it, err := db.NewIter(&pebble.IterOptions{LowerBound: from}) if err != nil { - return nil, fmt.Errorf("dedupe sweep: %w", err) + return nil, nil, fmt.Errorf("dedupe sweep: %w", err) } - b := db.NewBatch() - defer func() { _ = b.Close() }() - var next []byte - var expired, v0 int64 seen := 0 for valid := it.First(); valid; valid = it.Next() { if seen == sweepChunk { @@ -120,38 +141,66 @@ func (e *Embedded) sweepChunk(ctx context.Context, db *pebble.DB, from []byte, r break } seen++ - k := it.Key() - val := it.Value() - switch { - case !isCommit(val): + if sweepReason(it.Value(), now) != "" { + candidates = append(candidates, bytes.Clone(it.Key())) + } + } + if err := it.Close(); err != nil { + return nil, nil, fmt.Errorf("dedupe sweep: %w", err) + } + return candidates, next, nil +} + +// deleteSweepable re-reads each candidate and deletes those still sweepable, +// holding commitMu so no Commit lands between the re-read and the delete: a +// key re-committed after the unlocked read is never deleted with its new +// value. +func (e *Embedded) deleteSweepable(db *pebble.DB, candidates [][]byte) (expired, v0 int64, err error) { + e.commitMu.Lock() + defer e.commitMu.Unlock() + now := e.now() + b := db.NewBatch() + defer func() { _ = b.Close() }() + for _, k := range candidates { + val, closer, err := db.Get(k) + if errors.Is(err, pebble.ErrNotFound) { + continue + } + if err != nil { + return 0, 0, fmt.Errorf("dedupe sweep: %w", err) + } + reason := sweepReason(val, now) + _ = closer.Close() + switch reason { + case sweptVersion0: v0++ - case committedExpired(val, now): + case sweptExpired: expired++ default: continue } if err := b.Delete(k, nil); err != nil { - _ = it.Close() - return nil, fmt.Errorf("dedupe sweep: %w", err) + return 0, 0, fmt.Errorf("dedupe sweep: %w", err) } } - if err := it.Close(); err != nil { - return nil, fmt.Errorf("dedupe sweep: %w", err) - } - if e.sweepHook != nil { - e.sweepHook() + if e.sweepDeleteHook != nil { + e.sweepDeleteHook() } // NoSync: a delete lost to a crash is redone by the next pass. if err := b.Commit(pebble.NoSync); err != nil { - return nil, fmt.Errorf("dedupe sweep: %w", err) - } - res.Expired += int(expired) - res.Version0 += int(v0) - if expired > 0 { - sweptKeysCounter.Add(ctx, expired, metric.WithAttributes(attribute.String(sweptAttribute, sweptExpired))) + return 0, 0, fmt.Errorf("dedupe sweep: %w", err) } - if v0 > 0 { - sweptKeysCounter.Add(ctx, v0, metric.WithAttributes(attribute.String(sweptAttribute, sweptVersion0))) + return expired, v0, nil +} + +// sweepReason is why the sweep deletes a key holding val at now, or "" when +// it keeps it. +func sweepReason(val []byte, now time.Time) string { + switch { + case !isCommit(val): + return sweptVersion0 + case committedExpired(val, now): + return sweptExpired } - return next, nil + return "" } diff --git a/internal/dedupe/sweep_test.go b/internal/dedupe/sweep_test.go index da4ec9512..0c7ff46e9 100644 --- a/internal/dedupe/sweep_test.go +++ b/internal/dedupe/sweep_test.go @@ -100,28 +100,52 @@ func TestEmbedded_SweepDeletesExpiredAndVersionZeroKeys(t *testing.T) { assert.Equal(t, sweepResult{}, res, "a second pass finds nothing") } -// A Commit that arrives while a sweep chunk has read an expired key but not -// yet deleted it waits for the chunk, so the new commit is never deleted with -// the old value. Without the lock the Commit lands in the gap and the sweep -// then deletes it; the wait below only ever lets that pass, never fail. -func TestEmbedded_SweepNeverDeletesACommitLandingMidChunk(t *testing.T) { - t.Parallel() +// expiredAndClaimed commits id "e1" with an hour's retention, lets it expire +// and claims it again, for a test to commit mid-sweep. +func expiredAndClaimed(t *testing.T) (*Embedded, *Managed, []Claim) { + t.Helper() e := NewEmbedded(t.TempDir()) clock := newStepClock() SetClock(e, clock.now) m := switchedOn(t, e, "acme") commitIDs(t, m, time.Hour, "e1") clock.advance(2 * time.Hour) - claims, err := m.Reserve(context.Background(), []Key{{Table: "events", ID: "e1"}}, DefaultLease) require.NoError(t, err) require.Equal(t, Claimed, claims[0].Status, "expired: claimable again") + return e, m, claims +} + +// A key committed again after a sweep chunk read it as expired, but before +// the chunk re-read it, is kept: the re-read sees the new commit. +func TestEmbedded_SweepKeepsAKeyCommittedAfterItsRead(t *testing.T) { + t.Parallel() + e, m, claims := expiredAndClaimed(t) + var commitErr error + e.sweepScanHook = func() { commitErr = m.Commit(context.Background(), claims, time.Hour) } + res, err := e.sweep(context.Background(), e.db) + require.NoError(t, err) + require.NoError(t, commitErr) + assert.Equal(t, sweepResult{}, res) + + dup, err := mark(context.Background(), m, "e1") + require.NoError(t, err) + assert.True(t, dup, "the commit made after the read survived the sweep") +} + +// A Commit that arrives while a sweep chunk has re-read an expired key but +// not yet deleted it waits for the chunk, so the new commit is never deleted +// with the old value. Without the lock the Commit lands in the gap and the +// sweep then deletes it; the wait below only ever lets that pass, never fail. +func TestEmbedded_SweepNeverDeletesACommitLandingMidChunk(t *testing.T) { + t.Parallel() + e, m, claims := expiredAndClaimed(t) done := make(chan error, 1) - e.sweepHook = func() { + e.sweepDeleteHook = func() { go func() { done <- m.Commit(context.Background(), claims, time.Hour) }() select { - case <-done: - done <- nil + case err := <-done: + done <- err case <-time.After(50 * time.Millisecond): } } @@ -135,6 +159,49 @@ func TestEmbedded_SweepNeverDeletesACommitLandingMidChunk(t *testing.T) { assert.True(t, dup, "the commit made mid-chunk survived the sweep") } +// A chunk that starts over a long run of tombstones, as a tenant's version-0 +// block leaves until Pebble compacts it, reads through the run without the +// lock, so a Commit racing it waits for the chunk's re-reads and deletes +// alone. Sized for the race detector, which the unit suite runs under: there, +// a chunk holding the lock across this run kept a Commit waiting ~250 ms. +func TestEmbedded_SweepChunkOverTombstonesDoesNotHoldCommits(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + m := switchedOn(t, e, "acme") + b := e.db.NewBatch() + for i := range 300_000 { + require.NoError(t, b.Delete(fmt.Appendf(nil, "acme\x00%06d", i), nil)) + } + // A version-0 key after the run, so the chunk has one to delete. + require.NoError(t, b.Set([]byte("acme\x01"), make([]byte, 8), nil)) + require.NoError(t, b.Commit(pebble.NoSync)) + require.NoError(t, e.db.Flush()) + + type swept struct { + res sweepResult + err error + } + done := make(chan swept, 1) + go func() { + res, err := e.sweep(context.Background(), e.db) + done <- swept{res, err} + }() + var slowest time.Duration + for i := 0; ; i++ { + select { + case s := <-done: + require.NoError(t, s.err) + require.Equal(t, sweepResult{Version0: 1}, s.res) + assert.Less(t, slowest, 100*time.Millisecond, "the slowest Commit racing the sweep") + return + default: + } + start := time.Now() + commitIDs(t, m, 0, fmt.Sprintf("c%d", i)) + slowest = max(slowest, time.Since(start)) + } +} + // A retention is honoured on read before any sweep has run: the key is a // duplicate until the retention ends and claimable from that instant. func TestEmbedded_RetentionHonouredOnRead(t *testing.T) { From ffa4dd6875c168154e68fd4219601d7b5a380ea3 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Sat, 26 Sep 2026 04:54:27 -0400 Subject: [PATCH 08/11] docs(settings): name dedupe.retention in the last no-defaults claims dedupe.retention is the one config.json key the binary defaults (missing means "0", forever). The ingest handler's dedupe comment, the seed's doc comment, a registry test comment and the settings-directory CHANGELOG entry still said there were none. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01G6Cz4H5k1spJAPk5CZeGds --- CHANGELOG.md | 2 +- internal/api/ingest.go | 4 ++-- internal/settings/registry_test.go | 2 +- internal/settings/seed.go | 3 ++- 4 files changed, 6 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9c9feadbf..8019c2e34 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -24,7 +24,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Schema discovery captures each table's DDL, its columns' ordinals and default expressions, and the server version** (`internal/discovery/discovery.go`, `internal/testutil/testutil.go`): `Column` gains `DefaultExpression` and `Position` (both from a widened `system.columns` select), `TableSchema` gains `DDL` from `system.tables.create_table_query`, and `SchemaRegistry` gains `ServerVersion()` from a `SELECT version()` probe next to the existing `SELECT timezone()`. Groundwork for the native type layer, captured on the same refresh as the columns so a stale version cannot outlive the schemas it describes. That is a publication guarantee, not a same-server one: `chconn.Manager` resolves the connection per call, so a reload changing `clickhouse.addr` mid-refresh can still pair a version from one server with schemas from another — narrow, and self-correcting on the next refresh. `DDL` is `json:"-"` and does **not** appear in `/v1/ops/schema`: that endpoint marshals `TableSchema` straight to the client, and an external-engine table (S3, MySQL, PostgreSQL, Kafka) renders its wiring there unconditionally — endpoint, bucket or host, database, username, S3 access key id. ClickHouse masks the password itself as `[HIDDEN]` from ~23.9 (verified on 26.7.3), so the exposure is the topology rather than the secret — except on an older server, or one with `display_secrets_in_show_and_select` enabled. `position` and `default_expression` are additive fields in the response. A table listed in `system.tables` with no `system.columns` rows is skipped rather than published column-less, and both new queries fail the refresh on error exactly as `timezone()` and `system.columns` do — callers keep the prior cache and retry. -- **Settings-directory hot reload — boot loading, three reload triggers, and the config-key migration** (`internal/settings/` (new: `store.go`, `watch.go`, + tests), `internal/api/settings.go` (new, + tests), `internal/api/{router,ingest,structured_query}.go`, `internal/discovery/discovery.go`, `internal/config/config.go`, `cmd/wavehouse/main.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/settings-directory.mdx` (new — the hot-reloadable half of configuration gets its own page; `configuration.mdx` is boot config only); closes the loop [#500](https://github.com/Wave-RF/WaveHouse/pull/500) opened, tracked by [#48](https://github.com/Wave-RF/WaveHouse/issues/48)): the server now *consumes* the settings directory instead of only validating it. `settings.Store` owns the adopted snapshot: `settings.dir` / `WH_SETTINGS_DIR` is now **required**, boot validates and adopts the directory (missing or invalid refuses to start); a running instance then re-validates and re-adopts on any of three triggers — a **directory watch** (fsnotify on the directory, not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost; bursts debounce into one reload), **`SIGHUP`**, and **`POST /v1/ops/settings/reload`** (admin-gated; returns `{"adopted", "findings"}`, `200` adopted / `422` rejected) — all funneling through one serialized reload path. A reload that fails validation keeps the previous good snapshot (an operator mid-edit degrades to a log line, never a broken server); warnings don't block adoption, matching `wavehouse validate`. The tenant tunables **migrate out of boot config** into the directory's `config.json`: `dedupe.id_field` / `dedupe.require_id` (now with the per-table overrides under `dedupe.tables` that [#222](https://github.com/Wave-RF/WaveHouse/issues/222) asked for, resolved per record through the table → global cascade in one atomic snapshot read, so a reload lands at a record boundary and never mixes documents within one record), `query.default_max_rows` and `query.timestamp_bucket_seconds` (read per query), `schema.refresh_interval` (re-read after each tick, so a change applies from the next cycle), `stream.keepalive_interval` / `stream.keepalive_buckets` (a reload calls the new `Heartbeater.Reconfigure`, which rebuilds the keepalive wheel in place with every live subscriber carried over and re-times the running ticker) and `stream.gap_window_minutes` (the sweeper re-reads it every sweep), `mq.max_bytes_gb` (an after-adopt hook updates the tenant's ingest and dead-letter stream limits in place via `mq.Broker.SetMaxBytes` — shrinking below the buffered size backpressures until the sweeper purges it back under the limit, nothing is dropped), `dlq.enabled` with per-table overrides under `dlq.tables` (resolved by the ingest worker at the moment a poison row is isolated: on → park it on the tenant's dead-letter stream and ack; off → leave it unacked for redelivery, never dropped; a served tenant's DLQ stream and `GET /v1/ops/dlq/stats` always exist, so the switch is purely behavioral), the **ClickHouse wiring** (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout`: the new `chconn.Manager` is the one `driver.Conn` every consumer holds and swaps the connection behind it on reload — unconditionally, since the adopted settings are the authority and reachability already surfaces through schema discovery and `/readyz`; the replaced one closes after a `query_timeout` grace; the ingest worker, raw-SQL proxy, and schema registry read the HTTP target, timeout, and database per call), the **auth verifier wiring** (`auth.jwks_url` / `auth.role_claim`: the new `auth.Authenticator` swaps a whole verifier — key source plus its pinned algorithm allowlist — atomically per reload, unconditionally, so an unreachable JWKS fails closed until it can be fetched; `auth.Middleware` is gone — `Authenticator` is the one constructor), and the CORS allowlist (`cors.allowed_origins`, resolved per request). The corresponding YAML/env keys are **removed**: `server.cors_allowed_origins`, `query.default_max_rows`, `schema.refresh_interval`, `dedupe.enabled`, `dedupe.id_field`, `dedupe.require_id`, `stream.keepalive_interval`, `stream.keepalive_buckets`, `mq.gap_window_minutes`, `cache.timestamp_bucket_seconds`, `mq.max_bytes_gb`, `dlq.enabled`, `clickhouse.addr`, `clickhouse.http_port`, `clickhouse.http_scheme`, `clickhouse.database`, `clickhouse.username`, `clickhouse.query_timeout`, `auth.jwks_url`, `auth.role_claim` (and `WH_SERVER_CORS_ALLOWED_ORIGINS`, `WH_QUERY_DEFAULT_MAX_ROWS`, `WH_SCHEMA_REFRESH_INTERVAL`, `WH_DEDUPE_ENABLED`, `WH_DEDUPE_ID_FIELD`, `WH_DEDUPE_REQUIRE_ID`, `WH_STREAM_KEEPALIVE_INTERVAL`, `WH_STREAM_KEEPALIVE_BUCKETS`, `WH_MQ_GAP_WINDOW_MINUTES`, `WH_CACHE_TIMESTAMP_BUCKET_SECONDS`, `WH_MQ_MAX_BYTES_GB`, `WH_DLQ_ENABLED`, `WH_CH_ADDR`, `WH_CH_HTTP_PORT`, `WH_CH_HTTP_SCHEME`, `WH_CH_DATABASE`, `WH_CH_USERNAME`, `WH_CH_QUERY_TIMEOUT`, `WH_AUTH_JWKS_URL`, `WH_AUTH_ROLE_CLAIM`); the secrets — `clickhouse.password`, `auth.jwt_secret`, `auth.operator_key` — stay boot config on purpose (never in a tracked JSON file; combined with the adopted wiring on every reconnect, rotating one is a restart), and boot config is now **strict**: `config.Load` re-reads the YAML against the struct's tags and refuses to start naming every undeclared key, so a `dlq:` or `clickhouse: addr:` left behind can't be read, ignored, and believed; the binary carries **no compiled defaults** — every `config.json` key is required (validation names each missing one), so the adopted snapshot is what the files say, and once adopted it outlives its files (a deleted file or vanished directory is just a rejected reload). Defaults live in one checked-in seed directory (`internal/settings/seed/`, `go:embed`ded): the new **`wavehouse bootstrap [dir]`** writes it (refusing a non-empty directory, the `initdb` contract; the directory resolves exactly as it does for `validate` — the argument, else `WH_SETTINGS_DIR`, usage error with neither — so the two commands are interchangeable on one path and a bare `bootstrap` inside the container images seeds `/app/settings`), the dev `config.yaml` points at a gitignored `./settings` that `make dev` seeds from it, and the e2e fixture ships a copy. The container images ship **no** settings directory: `WH_SETTINGS_DIR` is preset to `/app/settings`, the operator mounts a directory there (`standalone.yaml` bind-mounts the checked-in `deployments/compose/settings/`), and a missing mount refuses to boot rather than running on defaults nobody chose. `dedupe.enabled` moves too: the new `dedupe.Managed` wraps the Pebble store and a `Store.AfterAdopt` hook opens or closes it after every adoption, so flipping the switch is a reload, not a restart (seen ids persist across an off/on cycle; a failed open on reload is logged and ingest fails closed with `500` until the next reload, since the files asked for dedupe — at boot it still refuses to start; a record caught in the instant of the flip is published un-deduped and counted by `wavehouse_ingest_dedupe_disabled_total` rather than failed, and the hook is registered before the boot apply so a reload can never leave the settings and the store out of step). The watcher reloads once as soon as its watch exists, closing the gap between the boot read and the watch — an edit landing in between (a ConfigMap update during a rolling restart) is adopted, not silently missed. `dedupe.enabled` / `WH_DEDUPE_ENABLED` are removed from boot config alongside the other keys. What stays in boot config is only what cannot change under a running process — resource sizing (`data_dir`, `cache.l1_max_cost`), the listeners, the observability exporters — and the secrets. The compose stack now bind-mounts a checked-in `deployments/compose/settings/` (the seed with `clickhouse.addr` pointed at the `clickhouse` service) instead of a volume seeded with `bootstrap`, so the quickstart is `up -d` again; the e2e orchestrator copies the fixture settings per run and patches the testcontainer's ClickHouse ports into `config.json`, since that wiring no longer has an env override. Every after-adopt hook (dedupe, keepalive wheel) is registered before the reload triggers start, so the watcher's first reload can never be missed by a hook. Consumers take functions, not values (`IngestHandler.DedupeSettings`, the structured-query handler's `defaultMaxRows` / `bucketSecs func() int`, the ingest worker's `dlqEnabled func(table) bool`, the sweeper's `gapWindow func() time.Duration`, `corsMiddleware`'s origins getter, `SchemaRegistry`'s database and refresh-interval sources, the query handlers' timeout sources), so `internal/api` stays testable without materializing settings directories. The settings directory is also the **runtime authority for access control and named pipes** (`internal/settings/store.go`, `internal/policy/source.go` (new), `internal/pipes/pipes.go`, `internal/api/{policy,pipes,router}.go`, `internal/stream/hub.go`, `internal/auth/auth.go`, `cmd/wavehouse/main.go`, `Makefile`, `deployments/compose/settings/{policies,roles}.json`, `clients/ts/src/settings.ts` (new); closes [#229](https://github.com/Wave-RF/WaveHouse/issues/229), [#33](https://github.com/Wave-RF/WaveHouse/issues/33), [#461](https://github.com/Wave-RF/WaveHouse/issues/461), [#514](https://github.com/Wave-RF/WaveHouse/issues/514), [#460](https://github.com/Wave-RF/WaveHouse/issues/460), [#363](https://github.com/Wave-RF/WaveHouse/issues/363); advances [#48](https://github.com/Wave-RF/WaveHouse/issues/48) and [#214](https://github.com/Wave-RF/WaveHouse/issues/214)): `roles.json`, `policies.json`, and `pipes.json` are adopted with `config.json` as one snapshot and re-adopted on the same three triggers, and **files are the only write path** — standalone, the operator edits them on the host; on WaveHouse Cloud the control plane writes them — so there is no stored copy that can skip validation: every adoption runs the current rules (strict decode rejecting unknown and duplicate keys, the full policy validation including the claim-template grammar, pipe name/SQL/parameter-type rules, and the cross-file check that every role a grant or `allowed_roles` names is declared in `roles.json`), and a rejected edit keeps the previous good policy and pipes in effect. `policies.json` is one policy document (`{}` = no policy, adopted fail-closed with a warning); `pipes.json` carries full definitions (`allowed_roles`, `parameters`, `description`), so a file-defined pipe is no longer admin-only by construction. Consumers read the adopted snapshot per request through `policy.Source` (a `func() *policy.Policy`; `settings.Store.Policy` in production, `policy.Static(p)` in tests) and `pipes.Source` (`settings.Store`; `pipes.Static(q...)` in tests), so a reload applies to the very next request, including the SSE hub's per-event policy read. `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, and pipe execution are unchanged; the operator key still passes the `/v1/ops/*` gate under no policy, now as the break-glass that inspects the policy and triggers `POST /v1/ops/settings/reload` after `policies.json` is fixed. The SDK gains `wh.settings.reload()` (`POST /v1/ops/settings/reload`, returning `{ adopted, findings }`). The compose stack's trial `public` policy moves into the bind-mounted `deployments/compose/settings/policies.json` + `roles.json`, and `make dev` copies the same two files into its seeded `./settings` so a fresh dev server works tokenless. **Removed** — the write endpoints `PUT /v1/ops/policy`, `PUT /v1/ops/pipes/{name}`, and `DELETE /v1/ops/pipes/{name}`; the NATS KV buckets `WAVEHOUSE_POLICY` and `WAVEHOUSE_PIPES` and their KV Watch sync (`internal/policy/store.go`, the pipes KV store); the boot-config keys `policy.file_path` / `WH_POLICY_FILE_PATH` and `pipes.dir` / `WH_PIPES_DIR` (a leftover `policy:` or `pipes:` YAML block now refuses boot by name, like the other moved keys) and the `.sql`-directory pipes bootstrap; `deployments/compose/dev-policy.yaml`; the SDK methods `wh.policy.set`, `wh.pipes.set`, and `wh.pipes.delete`; and the test helpers `policy.NewMemoryStore`, `pipes.NewMemoryStore`, and `testutil/natsjs.go`. +- **Settings-directory hot reload — boot loading, three reload triggers, and the config-key migration** (`internal/settings/` (new: `store.go`, `watch.go`, + tests), `internal/api/settings.go` (new, + tests), `internal/api/{router,ingest,structured_query}.go`, `internal/discovery/discovery.go`, `internal/config/config.go`, `cmd/wavehouse/main.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/settings-directory.mdx` (new — the hot-reloadable half of configuration gets its own page; `configuration.mdx` is boot config only); closes the loop [#500](https://github.com/Wave-RF/WaveHouse/pull/500) opened, tracked by [#48](https://github.com/Wave-RF/WaveHouse/issues/48)): the server now *consumes* the settings directory instead of only validating it. `settings.Store` owns the adopted snapshot: `settings.dir` / `WH_SETTINGS_DIR` is now **required**, boot validates and adopts the directory (missing or invalid refuses to start); a running instance then re-validates and re-adopts on any of three triggers — a **directory watch** (fsnotify on the directory, not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost; bursts debounce into one reload), **`SIGHUP`**, and **`POST /v1/ops/settings/reload`** (admin-gated; returns `{"adopted", "findings"}`, `200` adopted / `422` rejected) — all funneling through one serialized reload path. A reload that fails validation keeps the previous good snapshot (an operator mid-edit degrades to a log line, never a broken server); warnings don't block adoption, matching `wavehouse validate`. The tenant tunables **migrate out of boot config** into the directory's `config.json`: `dedupe.id_field` / `dedupe.require_id` (now with the per-table overrides under `dedupe.tables` that [#222](https://github.com/Wave-RF/WaveHouse/issues/222) asked for, resolved per record through the table → global cascade in one atomic snapshot read, so a reload lands at a record boundary and never mixes documents within one record), `query.default_max_rows` and `query.timestamp_bucket_seconds` (read per query), `schema.refresh_interval` (re-read after each tick, so a change applies from the next cycle), `stream.keepalive_interval` / `stream.keepalive_buckets` (a reload calls the new `Heartbeater.Reconfigure`, which rebuilds the keepalive wheel in place with every live subscriber carried over and re-times the running ticker) and `stream.gap_window_minutes` (the sweeper re-reads it every sweep), `mq.max_bytes_gb` (an after-adopt hook updates the tenant's ingest and dead-letter stream limits in place via `mq.Broker.SetMaxBytes` — shrinking below the buffered size backpressures until the sweeper purges it back under the limit, nothing is dropped), `dlq.enabled` with per-table overrides under `dlq.tables` (resolved by the ingest worker at the moment a poison row is isolated: on → park it on the tenant's dead-letter stream and ack; off → leave it unacked for redelivery, never dropped; a served tenant's DLQ stream and `GET /v1/ops/dlq/stats` always exist, so the switch is purely behavioral), the **ClickHouse wiring** (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout`: the new `chconn.Manager` is the one `driver.Conn` every consumer holds and swaps the connection behind it on reload — unconditionally, since the adopted settings are the authority and reachability already surfaces through schema discovery and `/readyz`; the replaced one closes after a `query_timeout` grace; the ingest worker, raw-SQL proxy, and schema registry read the HTTP target, timeout, and database per call), the **auth verifier wiring** (`auth.jwks_url` / `auth.role_claim`: the new `auth.Authenticator` swaps a whole verifier — key source plus its pinned algorithm allowlist — atomically per reload, unconditionally, so an unreachable JWKS fails closed until it can be fetched; `auth.Middleware` is gone — `Authenticator` is the one constructor), and the CORS allowlist (`cors.allowed_origins`, resolved per request). The corresponding YAML/env keys are **removed**: `server.cors_allowed_origins`, `query.default_max_rows`, `schema.refresh_interval`, `dedupe.enabled`, `dedupe.id_field`, `dedupe.require_id`, `stream.keepalive_interval`, `stream.keepalive_buckets`, `mq.gap_window_minutes`, `cache.timestamp_bucket_seconds`, `mq.max_bytes_gb`, `dlq.enabled`, `clickhouse.addr`, `clickhouse.http_port`, `clickhouse.http_scheme`, `clickhouse.database`, `clickhouse.username`, `clickhouse.query_timeout`, `auth.jwks_url`, `auth.role_claim` (and `WH_SERVER_CORS_ALLOWED_ORIGINS`, `WH_QUERY_DEFAULT_MAX_ROWS`, `WH_SCHEMA_REFRESH_INTERVAL`, `WH_DEDUPE_ENABLED`, `WH_DEDUPE_ID_FIELD`, `WH_DEDUPE_REQUIRE_ID`, `WH_STREAM_KEEPALIVE_INTERVAL`, `WH_STREAM_KEEPALIVE_BUCKETS`, `WH_MQ_GAP_WINDOW_MINUTES`, `WH_CACHE_TIMESTAMP_BUCKET_SECONDS`, `WH_MQ_MAX_BYTES_GB`, `WH_DLQ_ENABLED`, `WH_CH_ADDR`, `WH_CH_HTTP_PORT`, `WH_CH_HTTP_SCHEME`, `WH_CH_DATABASE`, `WH_CH_USERNAME`, `WH_CH_QUERY_TIMEOUT`, `WH_AUTH_JWKS_URL`, `WH_AUTH_ROLE_CLAIM`); the secrets — `clickhouse.password`, `auth.jwt_secret`, `auth.operator_key` — stay boot config on purpose (never in a tracked JSON file; combined with the adopted wiring on every reconnect, rotating one is a restart), and boot config is now **strict**: `config.Load` re-reads the YAML against the struct's tags and refuses to start naming every undeclared key, so a `dlq:` or `clickhouse: addr:` left behind can't be read, ignored, and believed; the binary carries **no compiled defaults** but one — every `config.json` key except `dedupe.retention` (missing means `"0"`, forever) is required (validation names each missing one), so the adopted snapshot is what the files say, and once adopted it outlives its files (a deleted file or vanished directory is just a rejected reload). Defaults live in one checked-in seed directory (`internal/settings/seed/`, `go:embed`ded): the new **`wavehouse bootstrap [dir]`** writes it (refusing a non-empty directory, the `initdb` contract; the directory resolves exactly as it does for `validate` — the argument, else `WH_SETTINGS_DIR`, usage error with neither — so the two commands are interchangeable on one path and a bare `bootstrap` inside the container images seeds `/app/settings`), the dev `config.yaml` points at a gitignored `./settings` that `make dev` seeds from it, and the e2e fixture ships a copy. The container images ship **no** settings directory: `WH_SETTINGS_DIR` is preset to `/app/settings`, the operator mounts a directory there (`standalone.yaml` bind-mounts the checked-in `deployments/compose/settings/`), and a missing mount refuses to boot rather than running on defaults nobody chose. `dedupe.enabled` moves too: the new `dedupe.Managed` wraps the Pebble store and a `Store.AfterAdopt` hook opens or closes it after every adoption, so flipping the switch is a reload, not a restart (seen ids persist across an off/on cycle; a failed open on reload is logged and ingest fails closed with `500` until the next reload, since the files asked for dedupe — at boot it still refuses to start; a record caught in the instant of the flip is published un-deduped and counted by `wavehouse_ingest_dedupe_disabled_total` rather than failed, and the hook is registered before the boot apply so a reload can never leave the settings and the store out of step). The watcher reloads once as soon as its watch exists, closing the gap between the boot read and the watch — an edit landing in between (a ConfigMap update during a rolling restart) is adopted, not silently missed. `dedupe.enabled` / `WH_DEDUPE_ENABLED` are removed from boot config alongside the other keys. What stays in boot config is only what cannot change under a running process — resource sizing (`data_dir`, `cache.l1_max_cost`), the listeners, the observability exporters — and the secrets. The compose stack now bind-mounts a checked-in `deployments/compose/settings/` (the seed with `clickhouse.addr` pointed at the `clickhouse` service) instead of a volume seeded with `bootstrap`, so the quickstart is `up -d` again; the e2e orchestrator copies the fixture settings per run and patches the testcontainer's ClickHouse ports into `config.json`, since that wiring no longer has an env override. Every after-adopt hook (dedupe, keepalive wheel) is registered before the reload triggers start, so the watcher's first reload can never be missed by a hook. Consumers take functions, not values (`IngestHandler.DedupeSettings`, the structured-query handler's `defaultMaxRows` / `bucketSecs func() int`, the ingest worker's `dlqEnabled func(table) bool`, the sweeper's `gapWindow func() time.Duration`, `corsMiddleware`'s origins getter, `SchemaRegistry`'s database and refresh-interval sources, the query handlers' timeout sources), so `internal/api` stays testable without materializing settings directories. The settings directory is also the **runtime authority for access control and named pipes** (`internal/settings/store.go`, `internal/policy/source.go` (new), `internal/pipes/pipes.go`, `internal/api/{policy,pipes,router}.go`, `internal/stream/hub.go`, `internal/auth/auth.go`, `cmd/wavehouse/main.go`, `Makefile`, `deployments/compose/settings/{policies,roles}.json`, `clients/ts/src/settings.ts` (new); closes [#229](https://github.com/Wave-RF/WaveHouse/issues/229), [#33](https://github.com/Wave-RF/WaveHouse/issues/33), [#461](https://github.com/Wave-RF/WaveHouse/issues/461), [#514](https://github.com/Wave-RF/WaveHouse/issues/514), [#460](https://github.com/Wave-RF/WaveHouse/issues/460), [#363](https://github.com/Wave-RF/WaveHouse/issues/363); advances [#48](https://github.com/Wave-RF/WaveHouse/issues/48) and [#214](https://github.com/Wave-RF/WaveHouse/issues/214)): `roles.json`, `policies.json`, and `pipes.json` are adopted with `config.json` as one snapshot and re-adopted on the same three triggers, and **files are the only write path** — standalone, the operator edits them on the host; on WaveHouse Cloud the control plane writes them — so there is no stored copy that can skip validation: every adoption runs the current rules (strict decode rejecting unknown and duplicate keys, the full policy validation including the claim-template grammar, pipe name/SQL/parameter-type rules, and the cross-file check that every role a grant or `allowed_roles` names is declared in `roles.json`), and a rejected edit keeps the previous good policy and pipes in effect. `policies.json` is one policy document (`{}` = no policy, adopted fail-closed with a warning); `pipes.json` carries full definitions (`allowed_roles`, `parameters`, `description`), so a file-defined pipe is no longer admin-only by construction. Consumers read the adopted snapshot per request through `policy.Source` (a `func() *policy.Policy`; `settings.Store.Policy` in production, `policy.Static(p)` in tests) and `pipes.Source` (`settings.Store`; `pipes.Static(q...)` in tests), so a reload applies to the very next request, including the SSE hub's per-event policy read. `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, and pipe execution are unchanged; the operator key still passes the `/v1/ops/*` gate under no policy, now as the break-glass that inspects the policy and triggers `POST /v1/ops/settings/reload` after `policies.json` is fixed. The SDK gains `wh.settings.reload()` (`POST /v1/ops/settings/reload`, returning `{ adopted, findings }`). The compose stack's trial `public` policy moves into the bind-mounted `deployments/compose/settings/policies.json` + `roles.json`, and `make dev` copies the same two files into its seeded `./settings` so a fresh dev server works tokenless. **Removed** — the write endpoints `PUT /v1/ops/policy`, `PUT /v1/ops/pipes/{name}`, and `DELETE /v1/ops/pipes/{name}`; the NATS KV buckets `WAVEHOUSE_POLICY` and `WAVEHOUSE_PIPES` and their KV Watch sync (`internal/policy/store.go`, the pipes KV store); the boot-config keys `policy.file_path` / `WH_POLICY_FILE_PATH` and `pipes.dir` / `WH_PIPES_DIR` (a leftover `policy:` or `pipes:` YAML block now refuses boot by name, like the other moved keys) and the `.sql`-directory pipes bootstrap; `deployments/compose/dev-policy.yaml`; the SDK methods `wh.policy.set`, `wh.pipes.set`, and `wh.pipes.delete`; and the test helpers `policy.NewMemoryStore`, `pipes.NewMemoryStore`, and `testutil/natsjs.go`. - **"Was this page helpful?" feedback widget on every docs page** (`docs/src/components/PageFeedback.astro` (new), `docs/src/components/Footer.astro`): a thumbs-up / thumbs-down vote below the page content, captured to PostHog as `docs_feedback` with `{ helpful, page }`. It renders from `Footer.astro`'s sidebar branch — the same indirection the Cloud CTA uses — rather than a per-page import or frontmatter flag, so every content page gets it automatically, including ones not written yet; it sits *below* the Cloud CTA on the pages that carry one, and splash pages (the homepage and 404) take the other footer branch and never render it. One vote per page per visitor: the choice is remembered in `localStorage` keyed by pathname, and a revisit renders the thanks message instead of re-prompting (storage is a nicety, not the record — a browser with storage disabled still votes). - **Settings-directory validation — `wavehouse validate [dir]`** (`internal/settings/` (new: `settings.go`, `validate.go`, `decode.go`, `finding.go`, + tests), `cmd/wavehouse/validate.go` (new, + tests), `cmd/wavehouse/main.go`): first piece of the file-based control plane (settings live in a directory of JSON documents — `roles.json`, `policies.json`, `pipes.json`, `config.json` — that a running instance will hot-reload; this change is validation-only — boot loading and reload wiring land separately). `settings.Validate(dir)` is the single gate every consumer of the directory runs: deliberately pure (no network, no ClickHouse — table/column existence stays with schema discovery, per Bring-Your-Own-Schema), and it collects **all** findings in one pass instead of failing on the first. Checks, layered: the directory holds exactly the four files (a missing file is an error — an empty document is `{}`, so absence always means deletion or a wrong path; any unexpected entry — file or directory — is an error so a typoed `polices.json` or a stray backup can't be silently ignored; dot-prefixed entries are the one carve-out, since erroring on vim swap files or the `..data` machinery Kubernetes ConfigMap mounts publish through would break hand editing and the cloud fan-out's mount pattern alike); strict JSON syntax (unknown fields rejected — the JSON form of the retired-config-key trap; empty/truncated files rejected, never read as an empty document; a leading UTF-8 byte order mark named as such instead of surfacing as a cryptic invalid-character error; a directory, unreadable file, or non-regular file (a FIFO would hang the read forever waiting for a writer; a stat gate rejects it — following symlinks, so Kubernetes ConfigMap mounts' symlink layout still passes) squatting on a settings filename named as the one real problem, not double-reported as "missing"; a top-level `null` rejected — the one well-formed document that decodes into a zero value without error, so it would silently read as "no settings"; trailing content rejected; duplicated object keys detected by a token-level pass, since `encoding/json` silently keeps the last copy); per-file shape rules (role names non-empty/unique, pipe names/SQL/param types, `config.json` bounds mirroring boot-config validation — its sections are the *tenant-owned* behavioral tunables (dedupe id_field/require_id plus per-table overrides under `dedupe.tables` — each entry overrides only the fields it names, resolving table → global → compiled default per field, so the effective id_field can never be empty — an explicit empty, whitespace-only, or whitespace-padded id_field is rejected at both levels, since an exact-match JSON key lookup would silently miss every row ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s shape, unblocked by the file design since table names are runtime-resolved like policy grants); query default_max_rows, schema refresh_interval, CORS origins); platform-owned knobs like the SSE keepalives deliberately stay boot config); and cross-file referential integrity (every role a policy grant, `default_role`/`admin_role`, or pipe allowlist references must be declared in `roles.json`; an empty role string in a grant or allowlist is named as such — it matches no request and authorizes nobody). Warnings don't invalidate: a grant scoping the admin role (an unconditional bypass — dead config), `default_role` = admin, and a `default` on a required pipe parameter are flagged but legal. An empty `policies.json` means no policy — fail closed, matching deleted-policy semantics — and draws a warning naming the total lockout, so it announces itself at validation time instead of one 403 at a time. The CLI (`cmd/wavehouse/validate.go`, following the `health` subcommand pattern) takes the directory as an argument or from `WH_SETTINGS_DIR`, prints findings, and exits 0/1/2 (valid/invalid/usage) so CI and operators can gate config changes before they reach a running instance. The dispatch in `main.go` also grows `help` and `version` subcommands, and an unknown command is now a usage error instead of silently falling through and starting the server (`wavehouse validat` booting a listener is not a typo anyone wants); each subcommand parses its arguments with a stdlib `flag.FlagSet`, so `wavehouse -h` prints command-specific help and a stray flag or argument is a usage error rather than being silently swallowed. `WH_SETTINGS_DIR` has a single authority: `config.EnvSettingsDir`, with a reflection test pinning the `settings.dir` struct tag to it. The directory's location joins boot config as `settings.dir` (`WH_SETTINGS_DIR`; `internal/config/config.go`, `config.yaml`, `docs/src/content/docs/configuration.mdx`) — boot-tier by necessity, since it's the pointer the reload machinery follows; no default, same silent-misconfiguration reasoning as `policy.file_path`. diff --git a/internal/api/ingest.go b/internal/api/ingest.go index 1189547a4..58126f942 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -704,8 +704,8 @@ func (h *IngestHandler) prepareRecord( // Optional deduplication. The dedupe settings resolve per record // from one snapshot (table override → global; the settings directory - // always states them, so no compiled fallback is needed), so a reload - // lands at a record boundary. A Deduplicator without a settings source is + // states them all but dedupe.retention, whose absence means "0"), so a + // reload lands at a record boundary. A Deduplicator without a settings source is // a wiring bug, not a mode — main wires both or neither. The id is claimed // in ingestWindow, once every record of the window is encoded, so nothing // but the publish can fail while the claim is held. diff --git a/internal/settings/registry_test.go b/internal/settings/registry_test.go index 3ac222ccb..a5abce849 100644 --- a/internal/settings/registry_test.go +++ b/internal/settings/registry_test.go @@ -79,7 +79,7 @@ func TestRegistry_ReloadWithWarningsAdopts(t *testing.T) { // TestOpen_RejectsInvalid pins the boot contract: an invalid directory yields // no Registry at all — there is no "store without a document" state and no -// compiled defaults to fall back on. +// compiled default for a required key to fall back on. func TestOpen_RejectsInvalid(t *testing.T) { t.Parallel() files := validFiles() diff --git a/internal/settings/seed.go b/internal/settings/seed.go index d9e979ae5..5315c1730 100644 --- a/internal/settings/seed.go +++ b/internal/settings/seed.go @@ -10,7 +10,8 @@ import ( // seedFS holds the starter settings directory: every file present, every // key set to its default. The checked-in seed/ directory is the ONE place -// defaults live — the binary has no compiled fallbacks. Its one consumer is +// defaults live — the binary's one compiled fallback is dedupe.retention +// (missing means "0", forever). Its one consumer is // this embed, so `wavehouse bootstrap` can write the directory anywhere // without a source tree; the container images ship no settings (the operator // mounts or seeds /app/settings), same as they ship no policy file. From 5e1ff2dd1b07448b165646d6a5f211514e141fbf Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Sat, 26 Sep 2026 08:33:30 -0400 Subject: [PATCH 09/11] test(dedupe): replace the sweep-lock timing assertion with structural ones MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit TestEmbedded_SweepChunkOverTombstonesDoesNotHoldCommits (e52b336c, this PR) raced a Commit against a sweep chunk over 300k tombstones and asserted the slowest attempt stayed under a 100ms wall-clock budget. That budget itself flaked under contention: 313ms measured when the whole internal/dedupe/... race suite ran, since test-unit runs packages in parallel under -race in make ci and a wall-clock bound moves with scheduler load. Replace it with two assertions load cannot move: a new sweepTouchHook (embedded.go, wired into deleteSweepable in sweep.go) tallies how many candidates the locked phase re-reads — asserted <= 1, the actual candidate count in this scenario, proving the locked phase's work is bounded by the candidates the unlocked read found sweepable, not by however many tombstones it silently stepped over inside Pebble to get there. A direct, non-blocking commitMu.TryLock() from within the existing sweepScanHook (fired after the unlocked read, before deleteSweepable's lock) proves the lock was free at that point without racing a goroutine or a clock at all: on the same goroutine that just did the read, TryLock fails instead of blocking if a regression left the lock held, so a pass is a direct proof, not an inference from a race won in time. Verified the new test actually catches the regression it guards against: temporarily wrapped sweepChunk's whole body (including the unlocked read) in e.commitMu.Lock()/Unlock() and dropped deleteSweepable's own lock to avoid a self-deadlock, simulating the pre-fix "sweep under the lock" design — the test failed exactly on the new TryLock assertion ("commitMu was free right after the unlocked read of 300k tombstones"), then reverted; `git diff` on sweep.go before adding the touch hook showed only the hook call, confirming a clean revert. GOTOOLCHAIN=go1.26.6 go test -race -count=20 -run Sweep ./internal/dedupe/ passed 20/20 while a concurrent `go test -race ./internal/...` ran in the background for contention (both processes exited 0). go build ./... and go vet -tags integration ./... are clean. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01G6Cz4H5k1spJAPk5CZeGds --- internal/dedupe/embedded.go | 7 ++++- internal/dedupe/sweep.go | 3 ++ internal/dedupe/sweep_test.go | 59 ++++++++++++++++++++--------------- 3 files changed, 42 insertions(+), 27 deletions(-) diff --git a/internal/dedupe/embedded.go b/internal/dedupe/embedded.go index 5eece3f3b..7cd60bd55 100644 --- a/internal/dedupe/embedded.go +++ b/internal/dedupe/embedded.go @@ -50,9 +50,14 @@ type Embedded struct { readHook func() error // sweepScanHook and sweepDeleteHook, when set, run in a sweep chunk: // between its unlocked read and its re-read, and between its re-read and - // its delete. A test races a Commit into each gap. + // its delete. A test races a Commit into each gap. sweepTouchHook, when + // set, runs once per candidate deleteSweepable re-reads under commitMu — + // a test tallies calls to pin that the locked phase's work is bounded by + // the candidate count, not by however many keys the unlocked read + // stepped over (silently, inside Pebble) to find them. sweepScanHook func() sweepDeleteHook func() + sweepTouchHook func() } // NewEmbedded returns the embedded implementation under dataDir. Nothing is diff --git a/internal/dedupe/sweep.go b/internal/dedupe/sweep.go index f1c4ca180..d5a49149e 100644 --- a/internal/dedupe/sweep.go +++ b/internal/dedupe/sweep.go @@ -162,6 +162,9 @@ func (e *Embedded) deleteSweepable(db *pebble.DB, candidates [][]byte) (expired, b := db.NewBatch() defer func() { _ = b.Close() }() for _, k := range candidates { + if e.sweepTouchHook != nil { + e.sweepTouchHook() + } val, closer, err := db.Get(k) if errors.Is(err, pebble.ErrNotFound) { continue diff --git a/internal/dedupe/sweep_test.go b/internal/dedupe/sweep_test.go index 0c7ff46e9..493e37459 100644 --- a/internal/dedupe/sweep_test.go +++ b/internal/dedupe/sweep_test.go @@ -161,45 +161,52 @@ func TestEmbedded_SweepNeverDeletesACommitLandingMidChunk(t *testing.T) { // A chunk that starts over a long run of tombstones, as a tenant's version-0 // block leaves until Pebble compacts it, reads through the run without the -// lock, so a Commit racing it waits for the chunk's re-reads and deletes -// alone. Sized for the race detector, which the unit suite runs under: there, -// a chunk holding the lock across this run kept a Commit waiting ~250 ms. +// lock: deleteSweepable's locked phase only ever touches the candidates the +// unlocked read found sweepable (one here), never the tombstones stepped +// over to find them, and commitMu is free the instant that read returns. +// +// This used to race a Commit against the sweep and assert its slowest +// attempt stayed under 100ms — a wall-clock budget the race detector alone +// could push past (250ms measured once), and one that moves further under +// make ci's parallel test-unit packages (313ms measured there once, this +// PR's e52b336c: the test that introduced it flaked in its own author's +// verification). A count and a direct, non-blocking TryLock don't move with +// scheduler contention, so they replace it. func TestEmbedded_SweepChunkOverTombstonesDoesNotHoldCommits(t *testing.T) { t.Parallel() e := NewEmbedded(t.TempDir()) - m := switchedOn(t, e, "acme") + switchedOn(t, e, "acme") b := e.db.NewBatch() for i := range 300_000 { require.NoError(t, b.Delete(fmt.Appendf(nil, "acme\x00%06d", i), nil)) } - // A version-0 key after the run, so the chunk has one to delete. + // A version-0 key after the run, so the chunk has exactly one candidate. require.NoError(t, b.Set([]byte("acme\x01"), make([]byte, 8), nil)) require.NoError(t, b.Commit(pebble.NoSync)) require.NoError(t, e.db.Flush()) - type swept struct { - res sweepResult - err error - } - done := make(chan swept, 1) - go func() { - res, err := e.sweep(context.Background(), e.db) - done <- swept{res, err} - }() - var slowest time.Duration - for i := 0; ; i++ { - select { - case s := <-done: - require.NoError(t, s.err) - require.Equal(t, sweepResult{Version0: 1}, s.res) - assert.Less(t, slowest, 100*time.Millisecond, "the slowest Commit racing the sweep") - return - default: + var touched int + e.sweepTouchHook = func() { touched++ } + var unlockedAfterScan bool + e.sweepScanHook = func() { + // Fires after the unlocked read, before deleteSweepable takes + // commitMu. TryLock needs no racing goroutine and no clock: on the + // same goroutine that just did the read, it fails instead of + // blocking if commitMu is already held — which a regression moving + // the read under the lock would leave it, self-deadlock included — + // so success here is a direct proof the read ran unlocked, not an + // inference from a race won in time. + if e.commitMu.TryLock() { + e.commitMu.Unlock() + unlockedAfterScan = true } - start := time.Now() - commitIDs(t, m, 0, fmt.Sprintf("c%d", i)) - slowest = max(slowest, time.Since(start)) } + + res, err := e.sweep(context.Background(), e.db) + require.NoError(t, err) + assert.Equal(t, sweepResult{Version0: 1}, res) + assert.True(t, unlockedAfterScan, "commitMu was free right after the unlocked read of 300k tombstones") + assert.LessOrEqual(t, touched, 1, "the locked phase touches only the sweepable candidates (1 here), not the tombstones the read stepped over to find them") } // A retention is honoured on read before any sweep has run: the key is a From c4147ecefefb0686057b16ab2a20a39c85b1a538 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Sat, 26 Sep 2026 08:53:41 -0400 Subject: [PATCH 10/11] test(dedupe): pin sweepCandidates' read as unlocked from inside its loop MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit TestEmbedded_SweepChunkOverTombstonesDoesNotHoldCommits claimed more than it checked. sweepTouchHook fired once per element of `candidates`, and the fixture has exactly one visible key, so `touched <= 1` could never fail: a full keyspace walk added inside the locked phase would still pass (R4). The TryLock ran in sweepScanHook, which fires AFTER sweepCandidates returns, so a regression that read under commitMu and released the lock just before that hook still passed (R3) — only "the whole chunk under one lock" was actually caught (R5). The 300k-tombstone fixture changed no verdict either way (a handful of tombstones behaves identically) while costing 1.65s alone / 3.1s under package load in a 15s-timeout package, and the comment recorded this test's own history (citing a commit that a squash would erase) instead of what it asserts. Drop sweepTouchHook (field, wiring, assertion). To pin "the read runs unlocked" for real, sweepCandidates now takes an onKey hook called once per key from inside its own loop, before evaluating it — wired through sweepChunk as e.sweepReadHook. Renamed TestEmbedded_SweepChunkOverTombstonesDoesNotHoldCommits to TestEmbedded_SweepReadRunsUnlocked: TryLock/Unlock from inside that hook, while the read is still running, so a lock held anywhere during the read is caught in the act rather than inferred from whether it was released before some later checkpoint. The fixture shrinks to a handful of tombstones ahead of the one live key, since the verdict never depended on the count. Comment cut to the one thing the test asserts; the old wall-clock/hook history belongs here instead. Verified by mutation, each applied then reverted (`git diff internal/dedupe/sweep.go` clean before the real change was made): - R3 (only the read under the lock, released right after): wrapped just the sweepCandidates call in sweepChunk with commitMu.Lock()/Unlock() — TestEmbedded_SweepReadRunsUnlocked failed ("commitMu must be free while sweepCandidates' read is running"). - R5 (the whole chunk under one lock): wrapped sweepChunk's body in commitMu.Lock()/defer Unlock() and dropped deleteSweepable's own lock to avoid a self-deadlock — ran ONLY the target test (not the package: TestEmbedded_SweepNeverDeletesACommitLandingMidChunk self-deadlocks under this mutation, racing a Commit against a sweep that never releases the lock) — failed with the same assertion. GOTOOLCHAIN=go1.26.6 go test -race -count=5 -run Sweep ./internal/dedupe/ passes (5/5, including TestEmbedded_SweepReadRunsUnlocked and every other Sweep-prefixed test); the full package (go test -race -count=1 ./internal/dedupe/...) passes too, now in ~5s rather than the prior fixture's ~57s at -count=20. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01G6Cz4H5k1spJAPk5CZeGds --- internal/dedupe/embedded.go | 11 ++++---- internal/dedupe/sweep.go | 13 +++++----- internal/dedupe/sweep_test.go | 47 ++++++++++++----------------------- 3 files changed, 28 insertions(+), 43 deletions(-) diff --git a/internal/dedupe/embedded.go b/internal/dedupe/embedded.go index 7cd60bd55..227564b71 100644 --- a/internal/dedupe/embedded.go +++ b/internal/dedupe/embedded.go @@ -50,14 +50,13 @@ type Embedded struct { readHook func() error // sweepScanHook and sweepDeleteHook, when set, run in a sweep chunk: // between its unlocked read and its re-read, and between its re-read and - // its delete. A test races a Commit into each gap. sweepTouchHook, when - // set, runs once per candidate deleteSweepable re-reads under commitMu — - // a test tallies calls to pin that the locked phase's work is bounded by - // the candidate count, not by however many keys the unlocked read - // stepped over (silently, inside Pebble) to find them. + // its delete. A test races a Commit into each gap. sweepReadHook, when + // set, runs once per key sweepCandidates' unlocked read visits, from + // inside its loop — a test TryLocks commitMu there to prove the read + // itself never holds it, not just the instant after it returns. sweepScanHook func() sweepDeleteHook func() - sweepTouchHook func() + sweepReadHook func() } // NewEmbedded returns the embedded implementation under dataDir. Nothing is diff --git a/internal/dedupe/sweep.go b/internal/dedupe/sweep.go index d5a49149e..25759f39b 100644 --- a/internal/dedupe/sweep.go +++ b/internal/dedupe/sweep.go @@ -104,7 +104,7 @@ func (e *Embedded) sweep(ctx context.Context, db *pebble.DB) (sweepResult, error // read and a run of them left by an earlier pass would otherwise hold every // Commit for its whole length. func (e *Embedded) sweepChunk(ctx context.Context, db *pebble.DB, from []byte, res *sweepResult) ([]byte, error) { - candidates, next, err := sweepCandidates(db, from, e.now()) + candidates, next, err := sweepCandidates(db, from, e.now(), e.sweepReadHook) if err != nil || len(candidates) == 0 { return next, err } @@ -128,14 +128,18 @@ func (e *Embedded) sweepChunk(ctx context.Context, db *pebble.DB, from []byte, r // sweepCandidates reads the next sweepChunk keys, starting at from, and // returns those sweepable at now and where the next chunk starts (nil at the -// end). -func sweepCandidates(db *pebble.DB, from []byte, now time.Time) (candidates [][]byte, next []byte, err error) { +// end). onKey, when non-nil, runs once per key visited, before it is +// evaluated — a test hook proving this read holds no lock while it runs. +func sweepCandidates(db *pebble.DB, from []byte, now time.Time, onKey func()) (candidates [][]byte, next []byte, err error) { it, err := db.NewIter(&pebble.IterOptions{LowerBound: from}) if err != nil { return nil, nil, fmt.Errorf("dedupe sweep: %w", err) } seen := 0 for valid := it.First(); valid; valid = it.Next() { + if onKey != nil { + onKey() + } if seen == sweepChunk { next = bytes.Clone(it.Key()) break @@ -162,9 +166,6 @@ func (e *Embedded) deleteSweepable(db *pebble.DB, candidates [][]byte) (expired, b := db.NewBatch() defer func() { _ = b.Close() }() for _, k := range candidates { - if e.sweepTouchHook != nil { - e.sweepTouchHook() - } val, closer, err := db.Get(k) if errors.Is(err, pebble.ErrNotFound) { continue diff --git a/internal/dedupe/sweep_test.go b/internal/dedupe/sweep_test.go index 493e37459..a0255e398 100644 --- a/internal/dedupe/sweep_test.go +++ b/internal/dedupe/sweep_test.go @@ -159,54 +159,39 @@ func TestEmbedded_SweepNeverDeletesACommitLandingMidChunk(t *testing.T) { assert.True(t, dup, "the commit made mid-chunk survived the sweep") } -// A chunk that starts over a long run of tombstones, as a tenant's version-0 -// block leaves until Pebble compacts it, reads through the run without the -// lock: deleteSweepable's locked phase only ever touches the candidates the -// unlocked read found sweepable (one here), never the tombstones stepped -// over to find them, and commitMu is free the instant that read returns. -// -// This used to race a Commit against the sweep and assert its slowest -// attempt stayed under 100ms — a wall-clock budget the race detector alone -// could push past (250ms measured once), and one that moves further under -// make ci's parallel test-unit packages (313ms measured there once, this -// PR's e52b336c: the test that introduced it flaked in its own author's -// verification). A count and a direct, non-blocking TryLock don't move with -// scheduler contention, so they replace it. -func TestEmbedded_SweepChunkOverTombstonesDoesNotHoldCommits(t *testing.T) { +// sweepCandidates' read never holds commitMu, over a fixture with a few +// tombstones ahead of the one live key it finds sweepable. +func TestEmbedded_SweepReadRunsUnlocked(t *testing.T) { t.Parallel() e := NewEmbedded(t.TempDir()) switchedOn(t, e, "acme") b := e.db.NewBatch() - for i := range 300_000 { - require.NoError(t, b.Delete(fmt.Appendf(nil, "acme\x00%06d", i), nil)) + for i := range 4 { + require.NoError(t, b.Delete(fmt.Appendf(nil, "acme\x00%02d", i), nil)) } - // A version-0 key after the run, so the chunk has exactly one candidate. require.NoError(t, b.Set([]byte("acme\x01"), make([]byte, 8), nil)) require.NoError(t, b.Commit(pebble.NoSync)) require.NoError(t, e.db.Flush()) - var touched int - e.sweepTouchHook = func() { touched++ } - var unlockedAfterScan bool - e.sweepScanHook = func() { - // Fires after the unlocked read, before deleteSweepable takes - // commitMu. TryLock needs no racing goroutine and no clock: on the - // same goroutine that just did the read, it fails instead of - // blocking if commitMu is already held — which a regression moving - // the read under the lock would leave it, self-deadlock included — - // so success here is a direct proof the read ran unlocked, not an - // inference from a race won in time. + var sawLocked bool + e.sweepReadHook = func() { + // TryLock from inside the still-running read needs no racing + // goroutine and no clock: on the same goroutine doing the read, it + // fails instead of blocking if commitMu is already held, so a + // regression that reads under the lock is caught while the read is + // still in progress — not inferred from a race won in time, and not + // missable by a lock released just before some later checkpoint. if e.commitMu.TryLock() { e.commitMu.Unlock() - unlockedAfterScan = true + } else { + sawLocked = true } } res, err := e.sweep(context.Background(), e.db) require.NoError(t, err) assert.Equal(t, sweepResult{Version0: 1}, res) - assert.True(t, unlockedAfterScan, "commitMu was free right after the unlocked read of 300k tombstones") - assert.LessOrEqual(t, touched, 1, "the locked phase touches only the sweepable candidates (1 here), not the tombstones the read stepped over to find them") + assert.False(t, sawLocked, "commitMu must be free while sweepCandidates' read is running") } // A retention is honoured on read before any sweep has run: the key is a From a76bbb9d86a6188b07413b9c6e64afe9d51b85e1 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Sat, 26 Sep 2026 09:00:32 -0400 Subject: [PATCH 11/11] test(dedupe): assert the sweep's read hook ran TestEmbedded_SweepReadRunsUnlocked asserted only that the hook never saw commitMu held, which also holds if the hook never fires: passing nil for the hook left it green. It now counts the visits and requires one; with the hook unwired it fails. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01G6Cz4H5k1spJAPk5CZeGds --- internal/dedupe/sweep_test.go | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/internal/dedupe/sweep_test.go b/internal/dedupe/sweep_test.go index a0255e398..ee87088e2 100644 --- a/internal/dedupe/sweep_test.go +++ b/internal/dedupe/sweep_test.go @@ -173,14 +173,11 @@ func TestEmbedded_SweepReadRunsUnlocked(t *testing.T) { require.NoError(t, b.Commit(pebble.NoSync)) require.NoError(t, e.db.Flush()) + var visits int var sawLocked bool e.sweepReadHook = func() { - // TryLock from inside the still-running read needs no racing - // goroutine and no clock: on the same goroutine doing the read, it - // fails instead of blocking if commitMu is already held, so a - // regression that reads under the lock is caught while the read is - // still in progress — not inferred from a race won in time, and not - // missable by a lock released just before some later checkpoint. + visits++ + // Non-blocking, on the reading goroutine: fails if the read holds commitMu. if e.commitMu.TryLock() { e.commitMu.Unlock() } else { @@ -191,6 +188,7 @@ func TestEmbedded_SweepReadRunsUnlocked(t *testing.T) { res, err := e.sweep(context.Background(), e.db) require.NoError(t, err) assert.Equal(t, sweepResult{Version0: 1}, res) + assert.Positive(t, visits, "the read hook ran") assert.False(t, sawLocked, "commitMu must be free while sweepCandidates' read is running") }