From d17754575c008a9a2b1a601d77083370730a95ef Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:21:40 -0400 Subject: [PATCH 01/70] build(chtypes): build, package and release against the chtypes SDK Pin the revision-6 26.6 artifact in chtypes.lock and fetch it with scripts/fetch-chtypes.sh. Every build is cgo: a glibc builder and a distroless/cc runtime that bakes the artifact, native-runner release builds for linux/amd64, linux/arm64 and darwin/arm64, and the setup-env action caches the artifact for the unit, integration and e2e jobs. golangci-lint moves to v2.13.2 (the next go directive needs it), which retires four nolint directives it no longer needs. Co-Authored-By: Claude Sonnet 5.5 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- .github/actionlint.yaml | 8 + .github/actions/setup-env/action.yml | 52 ++++ .github/workflows/README.md | 21 +- .github/workflows/ci.yml | 20 +- .github/workflows/goreleaser-validate.yml | 184 ++++++++++-- .github/workflows/publish-dev.yml | 219 ++++++++++---- .github/workflows/release.yml | 344 +++++++++++++++++----- .goreleaser.yaml | 188 +++++------- Makefile | 38 ++- chtypes.lock | 17 ++ deployments/Dockerfile | 46 ++- deployments/Dockerfile.goreleaser | 67 ++++- internal/app/discoveries.go | 2 +- internal/mq/external.go | 6 +- scripts/build.sh | 5 +- scripts/fetch-chtypes.sh | 98 ++++++ scripts/size.sh | 4 +- 17 files changed, 1002 insertions(+), 317 deletions(-) create mode 100644 chtypes.lock create mode 100755 scripts/fetch-chtypes.sh diff --git a/.github/actionlint.yaml b/.github/actionlint.yaml index 848bb6f3..8619c07b 100644 --- a/.github/actionlint.yaml +++ b/.github/actionlint.yaml @@ -7,6 +7,14 @@ # scope". Suppress only that exact message — a genuine scope typo (e.g. # `contents` → `conten`) still produces a different message and fails. Drop # this once actionlint ships the scope. +# +# nektos/act has the same gap and no equivalent escape hatch: `act` (0.2.89) +# refuses ci.yml outright with `Unknown Property code-quality` from its own +# schema validator, before running anything. That is the tool, not the +# workflow — GitHub accepts the scope. To dry-run or run ci.yml under act, +# copy the workflows to a scratch dir and strip the one `code-quality: write` +# line there. (act's default image also ships no Go, so `setup-env`'s +# `go version` step exits 127 under act regardless.) paths: "**/*.yml": ignore: diff --git a/.github/actions/setup-env/action.yml b/.github/actions/setup-env/action.yml index df58eef4..123d011c 100644 --- a/.github/actions/setup-env/action.yml +++ b/.github/actions/setup-env/action.yml @@ -44,6 +44,15 @@ # lockfile + astro.config.mjs. Speeds up warm `astro check` / # `astro build` — unchanged content skips the parse + transform # pipeline. See #132. +# 7. chtypes artifact cache (~/.cache/chtypes/artifacts/abi6) keyed on +# chtypes.lock — the pinned ClickHouse-version .so/.dylib the SDK +# dlopens. Measured for today's one pinned line (26.6, linux-amd64): +# 316 MiB of libchtypes.so on disk, ~65 MiB as a stored archive, and +# 85 MiB over the wire on a MISS (the upstream .tar.gz). Fetched via +# scripts/fetch-chtypes.sh (--frozen: refuses anything the lock +# doesn't name), which the CLI makes idempotent even on a cache hit +# (a lightweight manifest check, not a re-download). Go test jobs +# that dial chtypes need this; see .github/workflows/README.md. name: Setup CI environment description: Caches (restore + automatic post-job save) and toolchains for WaveHouse CI @@ -66,6 +75,9 @@ inputs: astro: description: "Cache the Astro content collections (docs check/build)" default: "false" + chtypes: + description: "Fetch + cache the chtypes artifact(s) pinned in chtypes.lock (Go test jobs that dial chtypes only)" + default: "false" # Exact-key cache-hit flags, for consumers that want to skip work on a # warm cache (e.g. docs-build's pnpm-store prune). Saves are NOT gated on @@ -178,6 +190,36 @@ runs: restore-keys: | golangci-${{ runner.os }}- + # chtypes artifact cache — keyed on chtypes.lock (not go.mod/go.sum): + # the pinned file+sha256 per platform/line is the only thing that + # changes its contents. runner.arch is in the key on principle (all CI + # runners are ubuntu-latest amd64 today; a future arm64 runner must not + # restore amd64 .so files into its search path). + # + # The SDK's default fetch dir is revision-scoped: + # ~/.cache/chtypes/artifacts/abi/-. The path and the key + # prefix both name the ABI revision (6 at SDK v0.4.0), so a cache saved + # by an older SDK is never restored into the search path. When the SDK's + # ABI revision changes, bump `abi6` in both places and re-lock (see + # docs/development.md). + - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + if: ${{ inputs.chtypes == 'true' }} + id: chtypes-cache + with: + path: ~/.cache/chtypes/artifacts/abi6 + key: chtypes-abi6-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('chtypes.lock') }} + restore-keys: | + chtypes-abi6-${{ runner.os }}-${{ runner.arch }}- + + # Runs on both cache hit and miss: scripts/fetch-chtypes.sh --frozen is + # cheap on a hit (a manifest check against chtypes.lock, not a + # re-download — see the script) and it is what turns a restored-but- + # unverified cache entry back into a hash-checked one every run. + - name: Fetch pinned chtypes artifact + if: ${{ inputs.chtypes == 'true' }} + shell: bash + run: scripts/fetch-chtypes.sh + # No actions/setup-go: it spends ~9s/job downloading + extracting a # toolchain the runner can already provide. The image's preinstalled # `go` + GOTOOLCHAIN=auto (the Go ≥1.21 default) resolve go.mod's @@ -189,6 +231,16 @@ runs: # this step makes that cost visible and the post-job save captures it. # Trade-off: setup-go's inline problem matchers (PR-file annotations # on compile errors) are gone; the log output is unchanged. + # + # On a chtypes job the cold cost is paid TWICE, and the second fetch is + # invisible because it happens inside the step above: `go run @ver` + # resolves its toolchain from the pinned CLI module, not from this + # repo's go.mod, so it can land on a different patch than the one this + # step materialises (measured on an ambient go1.24.13: go1.27.1 for the + # CLI, go1.27.0 for `go 1.27` here). Both are ~80 MB and both live in + # ~/go/pkg/mod/golang.org/toolchain, so gomod-v1 absorbs both after the + # first save. A `toolchain` line in go.mod naming the same patch the CLI + # resolves would collapse it to one — not worth pinning for ~7s. - name: Verify Go resolves go.mod's toolchain if: ${{ inputs.go == 'true' }} shell: bash diff --git a/.github/workflows/README.md b/.github/workflows/README.md index 470f82f8..66348434 100644 --- a/.github/workflows/README.md +++ b/.github/workflows/README.md @@ -40,7 +40,7 @@ Solid arrows are `needs` edges. Dotted arrows are **artifact polls** (invariant Break one of these knowingly or not at all. -1. **The aggregator job named `CI` is the only required status check.** The `main branch protection` ruleset requires `CI` and nothing else. The aggregator fails on any `failure`/`cancelled` need and treats `skipped` as passing — so path-filtered jobs (docs-only PRs skip the Go suites) and event-filtered jobs (title on pushes, deploys on PRs) never orphan the required check, and adding/renaming jobs never requires a ruleset edit. Consequence: every job that must gate merges **must be in the aggregator's `needs` list**. Two jobs are deliberately non-gating and excluded: `timing` (advisory wall-clock table) and `docs-preview` (the convenience Cloudflare preview deploy — `docs-build` already validates the build and *is* a need, so only the build gates; the preview deploy reports its own "Docs preview" check but, slow or failed, never delays or reds `CI`). +1. **The aggregator job named `CI` is the only required status check.** The `main branch protection` ruleset requires `CI` and nothing else. The aggregator fails on any `failure`/`cancelled` need and treats `skipped` as passing — so path-filtered jobs (docs-only PRs skip the Go suites) and event-filtered jobs (title on pushes, deploys on PRs) never orphan the required check, and adding/renaming jobs never requires a ruleset edit. Consequence: every job that must gate merges **must be in the aggregator's `needs` list**. Two jobs are deliberately non-gating and excluded: `timing` (advisory wall-clock table) and `docs-preview` (the convenience Cloudflare preview deploy — `docs-build` already validates the build and *is* a need, so only the build gates; the preview deploy reports its own "Docs preview" check but, slow or failed, never delays or reds `CI`). Outside `ci.yml` entirely, `.github/workflows/goreleaser-validate.yml` ("Release build validation") is advisory the same way — it runs on its own `pull_request` path filter and `workflow_dispatch`, never appears in the aggregator's `needs`, and making it gating would mean adding it there. 2. **A dedicated `coverage` job applies the consolidated gate, polling — not `needs`-ing — the suites.** Each suite (`unit`, `integration`, `e2e`) runs with `COV_DEFER=1` and uploads a `coverage-` fragment; the `coverage` job runs `make cov` (merge + every threshold gate) over all three — exactly like local `make ci`'s final step. Keeping it a separate job (not folded into e2e's tail) decouples the gate result from the e2e suite's pass/fail. Crucially it is `needs: changes` **only, not the suites**: a `needs` edge is a *scheduling* barrier — GitHub won't pick up a runner, check out, restore caches, or `pnpm install` until the needed jobs finish — so needing the suites would serialize this job's ~50s of setup onto the critical path after the last suite, for nothing (the setup doesn't depend on their results). Instead it starts at run creation, runs its setup in parallel with the suites, and blocks only at the merge by polling for the three fragments with [`scripts/ci/wait-artifact.sh`](../../scripts/ci/wait-artifact.sh) (fails fast if a producer concluded without producing). Tail on the critical path: ~10s, not ~50s. **The aggregator and `docs-deploy` must keep `coverage` *and* every suite in their `needs`** — the suites directly (a suite failure must red the gate even though `coverage` no longer needs them), and `coverage` (else a coverage-gate failure wouldn't block merge or a prod deploy). @@ -50,7 +50,7 @@ Break one of these knowingly or not at all. 5. **Trust domains.** Jobs that can reach deploy secrets (`docs-preview`, `docs-deploy`) check out **trusted `main`** and execute only files resolved from it — wrangler, the worker source, `wrangler.jsonc`, and any `scripts/ci/*.sh` they call ([#305](https://github.com/Wave-RF/WaveHouse/issues/305)). The only PR-derived input they touch is the static `docs-dist` artifact, consumed as data. Inline `run:` blocks in those jobs are acceptable (the workflow file itself is the reviewed surface); PR-tree *files* are not. Everything else (suites, lint, docs-build) runs the PR tree with no secrets beyond a read-mostly `GITHUB_TOKEN`. Fork PRs: secrets are absent and `docs-preview` skips itself. -6. **`ci.yml`'s caches are owned end-to-end by `setup-env`** ([.github/actions/setup-env](../actions/setup-env/action.yml)): each cache is a nested `actions/cache` step that restores inline and saves automatically at job end on an exact-key miss. No save steps in `ci.yml`. Trade-offs accepted: failed jobs don't save (restore-keys cushion the next run), and concurrent same-key misses produce benign "already exists" warnings. **Two cache steps live outside it**, both in `publish-dev.yml`, because that workflow doesn't use `setup-env` at all (it runs GoReleaser, not the test suites): a bare `actions/cache` owning the release build cache, and an `actions/cache/restore` that *reads* `gomod-v1` and owns nothing. Those two are the only `actions/cache*` steps outside the composite — keep it that way. A workflow that needs the shared module tree reads it restore-only; writing it belongs to the `ci.yml` jobs that run a full `go mod download`. +6. **`ci.yml`'s caches are owned end-to-end by `setup-env`** ([.github/actions/setup-env](../actions/setup-env/action.yml)): each cache is a nested `actions/cache` step that restores inline and saves automatically at job end on an exact-key miss. No save steps in `ci.yml`. Trade-offs accepted: failed jobs don't save (restore-keys cushion the next run), and concurrent same-key misses produce benign "already exists" warnings. **Two cache step definitions live outside it**, both in `publish-dev.yml`, because that workflow doesn't use `setup-env` at all (it runs GoReleaser, not the test suites): a bare `actions/cache` owning the release build cache, and an `actions/cache/restore` that *reads* `gomod-v1` and owns nothing. Both are instantiated once per matrix leg (`amd64`, `arm64` — each its own native `--single-target` build), so four cache-step runs total; still the only `actions/cache*` steps outside the composite — keep it that way. A workflow that needs the shared module tree reads it restore-only; writing it belongs to the `ci.yml` jobs that run a full `go mod download`. ## Coverage publishing @@ -80,15 +80,16 @@ Queue settings live in the `main branch protection` ruleset's `merge_queue` rule | pnpm store | `pnpm--` | any node job on miss | Store path resolved from pnpm at runtime. docs-build prunes before its save on a key rotation. | | Playwright Chromium | `playwright--` | docs-build | rehype-mermaid renders via headless Chrome at docs build. | | Astro content collections | `astro--` | lint / docs-build | Warm `astro check`/`build` skip unchanged content. | -| Go build objects (release) | `gobuild-v3--go-release-` | publish-dev (hand-rolled, not `setup-env`) | `~/.cache/go-build` from GoReleaser's 8-target cross-compile (~0.5 GB). Same family and key inputs as the CI flavors, `-release` suffix because cross-compiled objects share nothing with the native-only ones. Worth ≈2.5–7 min on every push to main (mean delta ≈4.8 min). | -| Go modules (release read) | `gomod-v1--` | nobody — **restore-only** | `publish-dev` reads `ci.yml`'s shared entry from `main`'s scope via `actions/cache/restore`, so its cross-compile isn't slowed by a cold module tree. No post-step save, so 0 GB of budget and no risk of a partial write to the shared key. | +| chtypes artifact | `chtypes-abi6---` | unit / integration / e2e via `setup-env` (shared) | `~/.cache/chtypes/artifacts/abi6` — the pinned ClickHouse-version `.so` the SDK dlopens; the SDK's default cache is per ABI revision (`abi/-`), and the revision is in both the path and the key prefix so an older SDK's cache is never restored. Measured for today's one pinned line (26.6, linux-amd64): **316 MiB on disk, ~0.06 GB stored**, and 85 MiB over the wire on a miss. Fetched by [`scripts/fetch-chtypes.sh`](../../scripts/fetch-chtypes.sh) (`--frozen`, refuses anything `chtypes.lock` doesn't name), which the CLI makes idempotent even on a cache hit — a manifest check, not a re-download. | +| Go build objects (release) | `gobuild-v3---go-release-` | publish-dev (hand-rolled, not `setup-env`) | `~/.cache/go-build` from each `publish-dev` leg's own native `goreleaser build --single-target` (~0.5 GB per arch). `runner.arch` is load-bearing in the key: both legs (`ubuntu-latest`, `ubuntu-24.04-arm`) report `runner.os == 'Linux'`, so without it they'd overwrite the same entry every run — a guaranteed miss on one of them, forever. Same family and key inputs as the CI flavors; `-release` suffix because these objects share nothing with the native test-build flavors. Budget is now **2 arches × 2 generations** of a native single-target cache, not the old 1 × 2 generations of a single 3-target cross-compile cache — still narrower than the pre-chtypes 8-target matrix (Windows/FreeBSD/darwin-amd64 dropped — chtypes publishes no artifact for any of them). | +| Go modules (release read) | `gomod-v1--` | nobody — **restore-only** | `publish-dev` reads `ci.yml`'s shared entry from `main`'s scope via `actions/cache/restore`, so its build isn't slowed by a cold module tree. No post-step save, so 0 GB of budget and no risk of a partial write to the shared key. | | CodeQL DB + deps | `codeql-dependencies-*`, `codeql-overlay-base-database-*` | GHAS default setup | **Not ours** — minted by GitHub's default CodeQL setup, not by any workflow in this repo, and not configurable here. ~0.4 GB. Listed so the budget arithmetic below is honest. | Deliberately **not** cached: `actions/setup-go`'s bundled cache (`cache: false` in `publish-dev.yml`, `release.yml` and `goreleaser-validate.yml`) — for different reasons per job. -It stores `~/go/pkg/mod` **and** `~/.cache/go-build` under one entry (~1 GB stored), keyed on the root `go.mod` — setup-go hashed `go.sum` through v6.2.0 and `go.mod` from v6.3.0, see [actions/setup-go#705](https://github.com/actions/setup-go/pull/705) — so roughly half of it re-stores the module tree `gomod-v1` already keeps once. `publish-dev.yml` opts out of that entry and caches the half that pays for itself on its own key (`gobuild-v3--go-release-`, ~0.5 GB): its GoReleaser step takes 36–246 s warm versus 401–446 s cold, so dropping the build objects outright would cost roughly 2.5–7 minutes on every push to main (mean delta ≈4.8 min across those runs). Those timings were measured with setup-go's bundled entry, which also held `~/go/pkg/mod` — so `publish-dev` additionally *restores* (never saves) `gomod-v1` from `main`'s scope, keeping the module tree warm too. Without that restore the job would re-download ~112 MB per push and land above the warm range this table quotes. +It stores `~/go/pkg/mod` **and** `~/.cache/go-build` under one entry (~1 GB stored), keyed on the root `go.mod` — setup-go hashed `go.sum` through v6.2.0 and `go.mod` from v6.3.0, see [actions/setup-go#705](https://github.com/actions/setup-go/pull/705) — so roughly half of it re-stores the module tree `gomod-v1` already keeps once. `publish-dev.yml` opts out of that entry and caches the half that pays for itself on its own key (`gobuild-v3---go-release-`, ~0.5 GB per arch): its GoReleaser step takes 36–246 s warm versus 401–446 s cold, so dropping the build objects outright would cost roughly 2.5–7 minutes on every push to main (mean delta ≈4.8 min across those runs). Those timings were measured with setup-go's bundled entry, which also held `~/go/pkg/mod` — so `publish-dev` additionally *restores* (never saves) `gomod-v1` from `main`'s scope, keeping the module tree warm too. Without that restore the job would re-download ~112 MB per push and land above the warm range this table quotes. -`release.yml` keeps the plain opt-out — no re-cache. After this change nothing mints a `setup-go-*` key at all, so turning its bundled cache back on would be a cold miss *and* a fresh ~1 GB save rather than a hit. What is warm is `publish-dev`'s `gobuild-v3--go-release-` entry, which a tag run could restore from the default branch's scope — but a tagged release is rare and not latency-sensitive, so it isn't worth a hand-rolled restore step. +`release.yml` keeps the plain opt-out — no re-cache. After this change nothing mints a `setup-go-*` key at all, so turning its bundled cache back on would be a cold miss *and* a fresh ~1 GB save rather than a hit. What is warm is `publish-dev`'s `gobuild-v3---go-release-` entries (one per arch), which a tag run could restore from the default branch's scope — but a tagged release is rare and not latency-sensitive, so it isn't worth a hand-rolled restore step. Re-enabling the bundled cache there would be strictly negative, not merely unhelpful: cache writes are scoped to the ref that made them, so a save from `refs/tags/v1.0.0` can never be read by `refs/tags/v1.0.1`, by `main`, or by a PR — only by a re-run of that same tag. It would be a ~1 GB entry per release that nothing but a retry can ever read. If release wall-clock ever does matter, the lever is `actions/cache/restore` on `publish-dev`'s key: restore-only, so it reads `main`'s warm entry and never writes a tag-scoped one. @@ -110,7 +111,7 @@ Include every family the rotation orphans, not just the renamed one — e.g. tur **Sizing policy — the repo cache budget is 10 GB, hard.** Past it GitHub LRU-evicts, so warm entries disappear mid-run and builds silently get slower. Budget for **two live generations**: a `go.mod`/`go.sum` or lockfile bump mints a whole new set while the previous one is still warm, so the steady state is ~2× a single generation. That is why `~/go/pkg/mod` is cached **once** (`gomod-v1`) rather than folded into each suffixed build cache — doing the latter stored the module tree five times over, five entries of ~0.9-1.2 GB each, ~5.2 GB per generation, and #438's 24-module bump pushed the repo to 10.53 GB ([#443](https://github.com/Wave-RF/WaveHouse/issues/443)). -Steady state after the split is roughly 5 GB of the 10 — two generations of `gomod-v1` + the five `gobuild-v3` flavors + the release build cache, plus the node-side caches and CodeQL. Before adding a cache or widening an existing `path:`, check the current footprint and confirm two generations still fit: +Steady state after the split is roughly 5 GB of the 10 — two generations of `gomod-v1` + the five `gobuild-v3` flavors + the release build cache (now two entries, one per arch, since the key carries `runner.arch`) — plus the node-side caches and CodeQL — plus ~0.13 GB for two generations of the `chtypes-abi6` artifact cache (~0.06 GB stored per generation, one pinned line today — the `.so` is 316 MiB on disk but compresses ~5x). Before adding a cache or widening an existing `path:`, check the current footprint and confirm two generations still fit: ```bash gh api repos/Wave-RF/WaveHouse/actions/cache/usage \ @@ -121,6 +122,12 @@ gh api repos/Wave-RF/WaveHouse/actions/caches --paginate \ Never add a per-job copy of content that is a pure function of a lockfile — key it once, unsuffixed, and let every job share it. +## chtypes artifacts + +`unit`, `integration` and `e2e` link the chtypes SDK (cgo dlopen of a per-ClickHouse-version `.so`/`.dylib`) and need the artifact for the line the test suite dials — today ClickHouse 26.6, matching `tests/integration/setup_test.go`'s pinned container. `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (job-level `env:` on all three) makes `typelayer.TestEngine` `t.Fatal()` if the artifact is missing instead of `t.Skip()`ing — CI must never quietly skip chtypes-backed tests. + +`chtypes.lock` (repo root) pins the exact file + sha256 per platform/line; `scripts/fetch-chtypes.sh` wraps the SDK's own CLI with `--frozen --lock chtypes.lock`, so a fetch here can only install what the lock names, never the rolling `artifacts` release. `setup-env`'s `chtypes: "true"` input (see [Cache inventory](#cache-inventory)) restores `~/.cache/chtypes/artifacts/abi6` and always re-runs the fetch script afterward — cheap on a hit (a manifest check, not a re-download) and what turns a restored-but-unverified cache entry back into a hash-checked one every run. A lock is specific to the SDK's ABI revision: a build from another revision is never selected, so after an SDK bump that changes the revision (6 at v0.4.0) `--frozen` fails with `CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED` until the lock is regenerated the same way, and the `abi6` path and key prefix in `setup-env` move with it. Widening the pinned line set is a two-step: `scripts/fetch-chtypes.sh ` locally to update `chtypes.lock`, then add the line to `LOCK_LINES` in that script. Two hash mismatches are possible and they behave differently — don't read one as the other. **Upstream republished a pinned line under a new sha256**: `--frozen` refuses the artifact the lock does not name, and both `Dockerfile.goreleaser`'s fetch and CI's cache-miss fetch fail, until `chtypes.lock` is regenerated per platform (`go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch 26.6 --lock chtypes.lock --platform `, once each for `darwin-arm64`, `linux-amd64`, `linux-arm64`, without `--frozen`) — and `goreleaser-validate.yml`'s image job, which exercises this same fetch on every PR touching `chtypes.lock` or the release workflows, is what surfaces a republish at PR time rather than mid-release. **The cache holds a library the current lock no longer names** (a re-lock landed, so the exact key missed and `restore-keys` handed back the previous generation): this does *not* fail — measured, the CLI reports `is present but hashes … (want …) — replacing` and re-downloads, then the post-job save mints the new generation. So a re-lock costs one cold fetch per Go job on the first run and nothing after. + ## Timing (steady state, full pipeline) The non-gating **Timing summary** job writes a per-job wall-clock table to every run's Summary page. Reference shape: diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index fa572e41..f8b3407e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -198,6 +198,11 @@ jobs: if: needs.changes.outputs.code == 'true' runs-on: ubuntu-latest timeout-minutes: 20 + # WAVEHOUSE_TEST_REQUIRE_CHTYPES=1: typelayer.TestEngine t.Fatal()s instead of + # t.Skip()ing when the pinned artifact isn't installed — CI must never + # silently skip the chtypes-backed tests it has the artifact for. + env: + WAVEHOUSE_TEST_REQUIRE_CHTYPES: "1" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: @@ -206,6 +211,7 @@ jobs: uses: ./.github/actions/setup-env with: go-cache-suffix: "-unit" + chtypes: "true" - name: Run Go unit tests + SDK vitest tests run: make test-unit test-ts COV_DEFER=1 - name: Upload coverage fragment @@ -223,6 +229,9 @@ jobs: if: needs.changes.outputs.code == 'true' runs-on: ubuntu-latest timeout-minutes: 20 + # See the unit job's comment on WAVEHOUSE_TEST_REQUIRE_CHTYPES. + env: + WAVEHOUSE_TEST_REQUIRE_CHTYPES: "1" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: @@ -237,7 +246,8 @@ jobs: uses: ./.github/actions/setup-env with: go-cache-suffix: "-integration" - node: "false" # pure Go + testcontainers ClickHouse + node: "false" # Go + testcontainers ClickHouse only + chtypes: "true" - name: Run Go integration tests run: make test-integration COV_DEFER=1 - name: Upload coverage fragment @@ -259,7 +269,12 @@ jobs: needs: changes if: needs.changes.outputs.code == 'true' runs-on: ubuntu-latest - timeout-minutes: 25 + # 35, not 25: a cold first push pays the cover-instrumented cgo build plus + # the chtypes artifact fetch on top of the suite. + timeout-minutes: 35 + # See the unit job's comment on WAVEHOUSE_TEST_REQUIRE_CHTYPES. + env: + WAVEHOUSE_TEST_REQUIRE_CHTYPES: "1" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: @@ -279,6 +294,7 @@ jobs: uses: ./.github/actions/setup-env with: go-cache-suffix: "-e2e-cov" + chtypes: "true" # `-j` builds the prereqs (build-ts ∥ build-cover) concurrently, # then runs the orchestrator: ClickHouse and Redis testcontainers + # the cover binary + the SDK vitest suite. diff --git a/.github/workflows/goreleaser-validate.yml b/.github/workflows/goreleaser-validate.yml index b09cd84e..ee4112c9 100644 --- a/.github/workflows/goreleaser-validate.yml +++ b/.github/workflows/goreleaser-validate.yml @@ -1,18 +1,47 @@ -name: GoReleaser snapshot validation +name: Release build validation -# Catches goreleaser config + Go-compile breakage at PR time, before -# it could block a real tag-time release. Snapshot mode implies -# --skip=publish, so nothing is pushed. +# PR-time proof that a real release would build. Nothing is published: +# `--snapshot` implies `--skip=publish`, and the image is built to +# `type=cacheonly`, so no registry is touched and no GitHub Release is made. # -# Path-filtered to goreleaser config / dockerfile changes only; -# go.mod / go.sum bumps are validated post-merge by publish-dev.yml -# running the full multi-target pipeline. `goreleaser build -# --single-target` exercises config parseability + Go compile for -# the host platform (release does not accept --single-target). +# It covers the three things a tag-time failure would otherwise block on: +# +# 1. `goreleaser check` — .goreleaser.yaml's schema. +# 2. a build per target — all THREE, each on its own native runner, which +# is the only way to compile them now that cgo +# rules out cross-compiling (darwin in particular: +# see the header of .goreleaser.yaml). The old +# `--single-target` job ran on linux/amd64 only and +# was blind to the other two by construction. +# 3. the image — the real multi-arch `docker buildx build` over +# deployments/Dockerfile.goreleaser with the +# snapshot binaries in the exact context layout +# release.yml assembles, including the chtypes +# artifact fetch from chtypes.lock. A stale lock, +# an unfetchable artifact or a Dockerfile mistake +# fails here rather than mid-release. +# +# Path filter: go.mod / go.sum are in it because a dependency bump is the +# realistic way a platform breaks (the darwin cgo C file that killed +# cross-compiling arrived with a prometheus/client_golang bump), and +# publish-dev.yml no longer builds darwin at all, so nothing else on main +# would catch it. chtypes.lock / fetch-chtypes.sh / _colors.sh are in it +# because the image build consumes all three — the assemble step copies +# _colors.sh into the context and the Dockerfile's fetch stage sources it, so +# an edit there can break the image with nothing else on a PR to catch it. +# +# NOT a required check: the `main branch protection` ruleset requires only +# `CI`, the aggregator in ci.yml. Making this gating means adding it to that +# aggregator's `needs`, which is a ci.yml change. on: pull_request: paths: - .goreleaser.yaml + - go.mod + - go.sum + - chtypes.lock + - scripts/fetch-chtypes.sh + - scripts/_colors.sh - deployments/Dockerfile.goreleaser - .github/workflows/release.yml - .github/workflows/publish-dev.yml @@ -27,34 +56,143 @@ concurrency: cancel-in-progress: true jobs: - validate: - name: Validate snapshot build + config: + name: Validate .goreleaser.yaml runs-on: ubuntu-latest - timeout-minutes: 10 + timeout-minutes: 5 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + # Runs on PR-authored code; nothing here needs authenticated git. + persist-credentials: false + + - uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 + with: + # Same v2 line as release.yml so this validates against the same + # goreleaser version a real release uses. + version: "~> v2" + args: check + + build: + name: Build ${{ matrix.goos }}/${{ matrix.goarch }} + runs-on: ${{ matrix.runner }} + timeout-minutes: 25 + strategy: + # Keep going: knowing that two of three targets are broken is more useful + # on a PR than the first failure alone. (release.yml is fail-fast for the + # opposite reason — there, a partial result is worthless.) + fail-fast: false + matrix: + # Must mirror release.yml's matrix and .goreleaser.yaml's declared + # goos/goarch. + include: + - goos: linux + goarch: amd64 + runner: ubuntu-latest + - goos: linux + goarch: arm64 + runner: ubuntu-24.04-arm + - goos: darwin + goarch: arm64 + runner: macos-latest steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: # goreleaser uses `git describe` to derive the snapshot # version; needs full tag history. fetch-depth: 0 - # Runs on PR-authored code; nothing here needs authenticated git. persist-credentials: false - uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0 with: go-version-file: "go.mod" - # Single-target build is fast enough that the bundled - # cache's post-step save costs more than a cold - # `go mod download`. + # A single-target build is fast enough that the bundled cache's + # post-step save costs more than a cold `go mod download` — and a + # cache saved from a PR's ref can only ever be read by that same PR. cache: false - - name: Run GoReleaser snapshot build + - name: Assert the runner is ${{ matrix.goos }}/${{ matrix.goarch }} + shell: bash + run: | + set -euo pipefail + host_os=$(go env GOOS) + host_arch=$(go env GOARCH) + echo "runner=${{ matrix.runner }} host=${host_os}/${host_arch}" + if [ "$host_os" != "${{ matrix.goos }}" ] || [ "$host_arch" != "${{ matrix.goarch }}" ]; then + echo "::error::${{ matrix.runner }} is ${host_os}/${host_arch} but the matrix says ${{ matrix.goos }}/${{ matrix.goarch }} — this leg is not validating what it claims to" + exit 1 + fi + + - name: Snapshot build uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 with: - # Same v2 line as release.yml so the snapshot validates - # against the same goreleaser version a real release uses. version: "~> v2" - # `build` (not `release`) so --single-target is accepted; - # release-only steps (archive / docker / checksums) are - # validated post-merge by publish-dev.yml. - args: build --snapshot --clean --single-target + # `--snapshot` because a PR head carries no release tag. + args: build --snapshot --clean --single-target --output dist/wavehouse + + - name: Describe the binary + shell: bash + run: | + set -euo pipefail + file dist/wavehouse + ls -l dist/wavehouse + + # Only the Linux binaries feed the image job; uploading the darwin one + # too would cost a transfer nothing reads. + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + if: matrix.goos == 'linux' + with: + name: snapshot-linux-${{ matrix.goarch }} + path: dist/wavehouse + if-no-files-found: error + retention-days: 1 + + image: + name: Build the multi-arch image + # A plain `needs` on a fail-fast: false matrix means ANY failing leg skips + # this job — including the darwin one, which contributes nothing here. That + # is the intended trade: if a target cannot even compile, the image result + # is noise, and the build job's own red is the finding. + needs: build + runs-on: ubuntu-latest + timeout-minutes: 25 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: snapshot-linux-* + path: staging + + - name: Assemble the image build context + shell: bash + run: | + set -euo pipefail + mkdir -p ctx/scripts + cp chtypes.lock ctx/ + cp scripts/fetch-chtypes.sh scripts/_colors.sh ctx/scripts/ + for goarch in amd64 arm64; do + mkdir -p "ctx/linux/${goarch}" + install -m 0755 "staging/snapshot-linux-${goarch}/wavehouse" \ + "ctx/linux/${goarch}/wavehouse" + done + find ctx -type f -printf '%M %s %p\n' + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0 + + # `type=cacheonly`: run the full build for both platforms — including the + # chtypes artifact fetch and sha256 verification against chtypes.lock — + # and throw the result away. No registry credentials, so this works on + # fork PRs too. + - name: Build (no push) + shell: bash + run: | + set -euo pipefail + docker buildx build \ + --platform linux/amd64,linux/arm64 \ + --file deployments/Dockerfile.goreleaser \ + --output type=cacheonly \ + ctx diff --git a/.github/workflows/publish-dev.yml b/.github/workflows/publish-dev.yml index c026cfc2..d8c46e10 100644 --- a/.github/workflows/publish-dev.yml +++ b/.github/workflows/publish-dev.yml @@ -4,15 +4,20 @@ name: Publish dev image # - :dev rolling pointer to the latest main commit # - :dev- immutable point-in-time reference # -# Reuses the goreleaser pipeline from release.yml. A synthetic local -# tag (v0.0.0-dev.) anchors goreleaser's .Version on -# HEAD; the tag is never pushed to origin. WAVEHOUSE_DEV=1 switches -# .goreleaser.yaml to dev image tags and suppresses the GitHub -# Release. +# Same shape as release.yml — native build jobs, then one job that assembles +# and pushes — because cgo makes cross-compiling impossible here (see the +# header of .goreleaser.yaml for the measurements). The difference is scope: +# a dev build publishes ONLY the image, so it builds only the two Linux +# targets that go into it. darwin/arm64 is not built here; its PR-time proof +# is goreleaser-validate.yml, which builds all three on a config or +# go.mod/go.sum change. # -# Real tagged releases (`v*`) flow through release.yml against the -# same .goreleaser.yaml without WAVEHOUSE_DEV set, producing `:vX.Y.Z` -# plus a moving channel pointer — `:latest` for a stable tag, but +# A synthetic local tag (v0.0.0-dev.) anchors goreleaser's .Version +# on HEAD; the tag is never pushed to origin. +# +# Real tagged releases (`v*`) flow through release.yml against the same +# .goreleaser.yaml, producing archives, a GitHub Release and `:vX.Y.Z` plus a +# moving channel pointer — `:latest` for a stable tag, but # `:alpha`/`:beta`/`:rc`/`:next` for a prerelease, which therefore never # touches `:latest`. Cleanup of old dev- tags is handled by # cleanup-ghcr.yml. @@ -23,11 +28,9 @@ on: branches: [main] workflow_dispatch: +# Least privilege by default; only `publish` is elevated. permissions: contents: read - packages: write - id-token: write # OIDC for the build-provenance attestation (Sigstore) - attestations: write # write the build-provenance attestation concurrency: # The :dev tag should track HEAD, so cancel any in-flight build @@ -38,18 +41,28 @@ concurrency: cancel-in-progress: true jobs: - publish: - name: Build + push :dev image - runs-on: ubuntu-latest - timeout-minutes: 30 + build: + name: Build linux/${{ matrix.goarch }} + runs-on: ${{ matrix.runner }} + timeout-minutes: 25 + strategy: + fail-fast: true + matrix: + # Only what the image needs. Must stay a subset of + # .goreleaser.yaml's declared goos/goarch. + include: + - goarch: amd64 + runner: ubuntu-latest + - goarch: arm64 + runner: ubuntu-24.04-arm steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - # goreleaser reads full git history for changelog + commit - # info. Matches release.yml. + # goreleaser reads full git history for commit info. Matches + # release.yml. fetch-depth: 0 - # Same reasoning as release.yml: a third-party action and a - # cross-compile run here, and nothing needs authenticated git. + # Same reasoning as release.yml: a third-party action and a full + # `go mod download` run here, and nothing needs authenticated git. persist-credentials: false - uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0 @@ -64,32 +77,30 @@ jobs: # `gomod-v1` — the duplication #443 is about. cache: false - # The other half is the reason this job is fast, so cache it on its - # own. GoReleaser cross-compiles 8 targets here (4 goos × 2 goarch), - # and the last 20 runs split cleanly on whether these objects were - # warm: GoReleaser finished in 36-246s with them, 401-446s without. - # (Measured on setup-go's bundled entry, which carried this same - # ~/.cache/go-build tree — this key is new here.) Dropping them - # outright would cost roughly 2.5-7 min per push to main (the envelope - # between those two clusters; mean delta ~4.8 min). + # The build objects are the reason this job is fast, so cache them on + # their own key. # - # Release-scoped suffix: these objects are cross-compiled for 8 - # GOOS/GOARCH pairs and share nothing with ci.yml's native-only - # flavors, so `-release` keeps both sides from restoring bytes the - # other can't use. Same `gobuild-v3` family and key inputs as - # setup-env's (see .github/workflows/README.md); ~0.5 GB rather than - # the ~1 GB the bundled cache held. + # `runner.arch` is new in the key and load-bearing: both legs of this + # matrix report `runner.os == 'Linux'`, so the pre-chtypes key + # (`gobuild-v3--go-release-…`) would have the amd64 and arm64 legs + # overwriting each other's entry every run — a guaranteed miss on one of + # them, forever. Still the `gobuild-v3` family and the same key inputs as + # setup-env's (see .github/workflows/README.md); `-release` keeps these + # from restoring into ci.yml's native test-build flavors, which carry + # different tags and a coverage instrumentation variant. - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ~/.cache/go-build - key: gobuild-v3-${{ runner.os }}-go-release-${{ hashFiles('**/go.mod', '**/go.sum') }} + key: gobuild-v3-${{ runner.os }}-${{ runner.arch }}-go-release-${{ hashFiles('**/go.mod', '**/go.sum') }} restore-keys: | - gobuild-v3-${{ runner.os }}-go-release- + gobuild-v3-${{ runner.os }}-${{ runner.arch }}-go-release- # And read — never write — ci.yml's shared module tree, so dropping # setup-go's bundled cache doesn't leave this job re-downloading ~112 MB # of modules on every push. This workflow runs on main, the same scope - # ci.yml saves `gomod-v1` into, so the entry is there to hit. + # ci.yml saves `gomod-v1` into, so the entry is there to hit. No arch in + # this key: the module cache is downloaded source, identical on both + # legs, so the arm64 leg reads the entry amd64 jobs wrote. # # restore, not cache: a full actions/cache would add a post-step save, # and this job has no business writing the entry every ci.yml Go job @@ -103,11 +114,91 @@ jobs: restore-keys: | gomod-v1-${{ runner.os }}- - # dockers_v2 builds the linux/amd64+arm64 manifest via `docker buildx`; - # the GitHub-hosted runner's default docker driver can't build - # multi-platform images, so create a docker-container builder. No QEMU - # needed — Dockerfile.goreleaser stages are $BUILDPLATFORM-pinned / - # COPY-only by design. + # Before the build: if `ubuntu-latest` ever moves to arm64 this would + # otherwise publish two identical arm64 binaries under different names. + - name: Assert the runner is linux/${{ matrix.goarch }} + shell: bash + run: | + set -euo pipefail + host_os=$(go env GOOS) + host_arch=$(go env GOARCH) + echo "runner=${{ matrix.runner }} host=${host_os}/${host_arch}" + if [ "$host_os" != linux ] || [ "$host_arch" != "${{ matrix.goarch }}" ]; then + echo "::error::${{ matrix.runner }} is ${host_os}/${host_arch} but the matrix says linux/${{ matrix.goarch }} — cgo cannot cross-compile these targets, so this run would ship the wrong binary" + exit 1 + fi + + - name: Create synthetic dev tag + # `goreleaser build` needs a tag on HEAD to derive .Version. Created + # locally with --force (retries on the same commit don't collide) and + # never pushed to origin. The pushed image tags are computed in the + # publish job below. + run: git tag --force "v0.0.0-dev.${GITHUB_SHA::7}" + + - name: Build + uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 + with: + # Same v2 line as release.yml so dev exercises the same + # goreleaser version a real release will. + version: "~> v2" + args: build --clean --single-target --output dist/wavehouse + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: wavehouse-linux-${{ matrix.goarch }} + path: dist/wavehouse + if-no-files-found: error + retention-days: 1 + + publish: + name: Push :dev image + needs: build + runs-on: ubuntu-latest + timeout-minutes: 30 + permissions: + # A job-level block REPLACES the workflow-level one outright, so + # `contents: read` has to be restated here or `actions/checkout` has no + # token to read the repo with. + contents: read + packages: write + id-token: write # OIDC for the build-provenance attestation (Sigstore) + attestations: write # write the build-provenance attestation + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: wavehouse-linux-* + path: staging + + - name: Assemble the image build context + shell: bash + run: | + set -euo pipefail + mkdir -p ctx/scripts + # Dockerfile.goreleaser's chtypes-fetch stage needs these three at + # their normal repo-relative paths. + cp chtypes.lock ctx/ + cp scripts/fetch-chtypes.sh scripts/_colors.sh ctx/scripts/ + + for goarch in amd64 arm64; do + bin="staging/wavehouse-linux-${goarch}/wavehouse" + if [ ! -f "$bin" ]; then + echo "::error::missing build artifact for linux/${goarch} (${bin})" + exit 1 + fi + # actions/upload-artifact's zip carries no unix mode bits, so the + # executable bit does not survive the round trip. + mkdir -p "ctx/linux/${goarch}" + install -m 0755 "$bin" "ctx/linux/${goarch}/wavehouse" + done + find ctx -type f -printf '%M %s %p\n' + + # The default docker driver can't build multi-platform images, so create + # a docker-container builder. No QEMU needed — Dockerfile.goreleaser's + # stages are $BUILDPLATFORM-pinned or COPY-only by design. - name: Set up Docker Buildx uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0 @@ -118,30 +209,36 @@ jobs: username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} - - name: Create synthetic dev tag - # `goreleaser release` requires a tag on HEAD to derive - # .Version. Created locally with --force (retries on the - # same commit don't collide) and never pushed to origin. - # The pushed image tags come from .goreleaser.yaml's - # WAVEHOUSE_DEV branch (`dev-` + `dev`). - run: git tag --force "v0.0.0-dev.${GITHUB_SHA::7}" - - - name: Run GoReleaser - uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 + # `:dev-` matches cleanup-ghcr.yml's `^dev-[0-9a-f]+$` regex. + # `:dev` is written as a literal rather than derived from a channel + # variable: a push to main that forgot to set something must never be + # able to publish `:latest`. + - name: Build and push the multi-arch dev image + shell: bash env: - # Switches .goreleaser.yaml into dev mode: dev/dev- - # image tags, dev OCI labels, GitHub Release suppressed. - WAVEHOUSE_DEV: "1" - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - with: - # Same v2 line as release.yml so dev exercises the same - # goreleaser version a real release will. - version: "~> v2" - args: release --clean + IMAGE: ghcr.io/wave-rf/wavehouse + run: | + set -euo pipefail + version="0.0.0-dev.${GITHUB_SHA::7}" + created="$(date -u +%Y-%m-%dT%H:%M:%SZ)" + docker buildx build \ + --platform linux/amd64,linux/arm64 \ + --file deployments/Dockerfile.goreleaser \ + --tag "${IMAGE}:dev-${GITHUB_SHA}" \ + --tag "${IMAGE}:dev" \ + --label "org.opencontainers.image.title=WaveHouse (dev)" \ + --label "org.opencontainers.image.description=Schema-aware real-time API gateway for ClickHouse — rolling dev build from main" \ + --label "org.opencontainers.image.url=${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}" \ + --label "org.opencontainers.image.source=${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}" \ + --label "org.opencontainers.image.version=${version}" \ + --label "org.opencontainers.image.revision=${GITHUB_SHA}" \ + --label "org.opencontainers.image.created=${created}" \ + --push \ + ctx # Attest the rolling dev image (multi-arch manifest-list digest) and store # the attestation alongside it in GHCR. Free for public repos via Sigstore. - # Dev binaries aren't distributed (the GitHub Release is suppressed), so + # Dev binaries aren't distributed (no GitHub Release, no archives), so # only the image is attested. Mirrors release.yml. - name: Resolve pushed dev image digest id: image diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index dd6fa9c4..0672b07b 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1,15 +1,30 @@ name: Release +# Shape: three NATIVE build jobs, one publish job. +# +# cgo made cross-compiling impossible for darwin (measured — see the header of +# .goreleaser.yaml: a dependency's own darwin C file needs Mach headers no +# Linux runner can supply), and the GoReleaser features that solve it — +# split/merge, `builder: prebuilt` — are Pro-only, with no `--skip=build` in +# OSS. So GoReleaser compiles, once per runner, and this workflow does the +# assembling: archives, checksums, GitHub Release, and the multi-arch GHCR +# image built straight from `deployments/Dockerfile.goreleaser` (the same +# `//wavehouse` context layout GoReleaser's dockers_v2 used to hand +# it, so that file is unchanged). +# +# Every runner label here is free for public repos: ubuntu-latest, +# ubuntu-24.04-arm, macos-latest. + on: push: tags: - "v*" +# Least privilege by default. Only `publish` is elevated — the three build +# jobs compile a tag's code on three runners and must not be able to write a +# release, push a package, or mint an attestation. permissions: - contents: write # create the GitHub Release (goreleaser) - packages: write # push the image to GHCR - id-token: write # OIDC for build-provenance attestations (Sigstore) - attestations: write # write the build-provenance attestations + contents: read concurrency: # Per-tag, never cancelling: two runs of the SAME tag (a re-run after a @@ -19,54 +34,195 @@ concurrency: cancel-in-progress: false jobs: - release: - name: Release - runs-on: ubuntu-latest - timeout-minutes: 30 + build: + name: Build ${{ matrix.goos }}/${{ matrix.goarch }} + runs-on: ${{ matrix.runner }} + timeout-minutes: 25 + strategy: + # A release is all-or-nothing: there is no such thing as publishing two + # of the three platforms, so stop the others as soon as one fails. + fail-fast: true + matrix: + # Must mirror .goreleaser.yaml's `builds[0].goos/goarch/ignore`. + # `--single-target` compiles the RUNNER's platform and ignores that + # list, so this matrix is what actually decides what gets built — the + # guard step below is what keeps the two from drifting apart. + include: + - goos: linux + goarch: amd64 + runner: ubuntu-latest + - goos: linux + goarch: arm64 + runner: ubuntu-24.04-arm + - goos: darwin + goarch: arm64 + runner: macos-latest steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: + # goreleaser derives .Version from the tag and .ShortCommit from the + # commit; `git describe` needs the tag history. fetch-depth: 0 - # The most privileged job here: contents+packages+attestations write, - # running a third-party action, a downloaded goreleaser binary, a full - # `go mod download` + cross-compile, and a Docker build — any one of - # which could read a persisted token out of .git/config and push. No - # authenticated git is needed: GoReleaser talks to the API via - # GITHUB_TOKEN, `git describe` is local, and the fetch is already done. + # Nothing here needs authenticated git: `git describe` is local and + # this job pushes nothing. A downloaded goreleaser binary and a full + # `go mod download` run before anything else, and a persisted token + # in .git/config would be readable by both. persist-credentials: false - uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0 with: go-version-file: "go.mod" - # Opting out keeps a ~1 GB archive — the module tree plus this - # job's cross-compile objects — out of the 10 GB repo budget. - # (That entry is keyed on the root go.mod, not go.sum: setup-go - # hashed go.sum through v6.2.0 and go.mod from v6.3.0 — - # actions/setup-go#705.) - # - # Turning it back on here would be strictly negative. Nothing - # mints that key any more — publish-dev.yml and - # goreleaser-validate.yml opt out too, and ci.yml runs no - # setup-go — so the restore is a guaranteed miss. Worse, cache - # writes are scoped to the ref that made them: a save from - # refs/tags/v1.0.0 can never be read by refs/tags/v1.0.1, by main, - # or by a PR — only by a re-run of that same tag. It would be a - # ~1 GB entry per release that nothing but a retry can ever read. - # - # What IS warm is publish-dev's gobuild-v3--go-release- entry, - # restorable from the default branch's scope. A tagged release is - # rare and not latency-sensitive, so it isn't worth a hand-rolled - # restore step here — but that's the lever if it ever is. - # - # Same cache: false as publish-dev.yml and goreleaser-validate.yml; - # different follow-up (publish-dev re-caches the useful half). #443. + # No cache. Cache writes are scoped to the ref that made them, so a + # save from refs/tags/v1.0.0 can never be read by refs/tags/v1.0.1, + # by main, or by a PR — only by a re-run of that same tag. It would + # be a ~1 GB entry per release that nothing but a retry can read. + # A tagged release is rare and not latency-sensitive. (publish-dev + # caches instead, on main's scope, where it is read back. #443.) cache: false - # dockers_v2 builds the linux/amd64+arm64 manifest via `docker buildx`; - # the GitHub-hosted runner's default docker driver can't build + # Before the build, not after: if a runner label ever changes + # architecture under us (ubuntu-latest moving to arm64 would do it), + # `--single-target` would silently produce the wrong binary and this + # workflow would publish it under the matrix's name. + - name: Assert the runner is ${{ matrix.goos }}/${{ matrix.goarch }} + shell: bash + run: | + set -euo pipefail + host_os=$(go env GOOS) + host_arch=$(go env GOARCH) + echo "runner=${{ matrix.runner }} host=${host_os}/${host_arch}" + if [ "$host_os" != "${{ matrix.goos }}" ] || [ "$host_arch" != "${{ matrix.goarch }}" ]; then + echo "::error::${{ matrix.runner }} is ${host_os}/${host_arch} but the matrix says ${{ matrix.goos }}/${{ matrix.goarch }} — cgo cannot cross-compile these targets, so this run would ship the wrong binary" + exit 1 + fi + + - name: Build + uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 + with: + # Pin to the v2 line so a v3 release breaks loudly instead of + # silently changing release behavior. + version: "~> v2" + # `build`, not `release`: this job only compiles. `--output` copies + # the binary to a fixed path so nothing downstream has to know + # GoReleaser's per-target dist directory name (`wavehouse_darwin_ + # arm64_v8.0` and friends — the GOARM64 suffix is not obvious). + args: build --clean --single-target --output dist/wavehouse + + - name: Describe the binary + shell: bash + run: | + set -euo pipefail + file dist/wavehouse + ls -l dist/wavehouse + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: wavehouse-${{ matrix.goos }}-${{ matrix.goarch }} + path: dist/wavehouse + if-no-files-found: error + # An intra-run hand-off to `publish`, not a deliverable — the + # deliverables are the release assets and the image. + retention-days: 1 + + publish: + name: Publish + needs: build + runs-on: ubuntu-latest + timeout-minutes: 30 + permissions: + contents: write # create the GitHub Release + packages: write # push the image to GHCR + id-token: write # OIDC for build-provenance attestations (Sigstore) + attestations: write # write the build-provenance attestations + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + # `git describe` picks the previous v* tag for the release notes. + fetch-depth: 0 + # The most privileged job here: contents+packages+attestations write, + # running a Docker build that fetches from the network — which could + # read a persisted token out of .git/config and push. No + # authenticated git is needed: `gh` reaches the API via GH_TOKEN, + # `git describe` is local, and the fetch is already done. + persist-credentials: false + + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: wavehouse-* + path: staging + + # Resolve the moving GHCR tag BEFORE anything is published, so a + # malformed tag fails here rather than after a multi-arch image push. + # The script rejects anything that isn't semver-shaped, so a typo can't + # fall through to `latest`. It is also the single source of truth for + # "is this a prerelease" below — `latest` means stable, by definition — + # rather than a second semver parser that could disagree with it. + - name: Resolve release channel + shell: bash + run: | + set -euo pipefail + channel="$(scripts/ci/release-channel.sh "$GITHUB_REF_NAME")" + echo "WAVEHOUSE_CHANNEL=${channel}" >> "$GITHUB_ENV" + echo "::notice::${GITHUB_REF_NAME} publishes ghcr.io/wave-rf/wavehouse:${GITHUB_REF_NAME} + :${channel}" + + - name: Assemble archives, checksums, and the image build context + shell: bash + run: | + set -euo pipefail + mkdir -p dist ctx/scripts + + # Dockerfile.goreleaser's chtypes-fetch stage needs these three at + # their normal repo-relative paths — the same files GoReleaser's + # dockers_v2.extra_files used to copy into its temp context. + cp chtypes.lock ctx/ + cp scripts/fetch-chtypes.sh scripts/_colors.sh ctx/scripts/ + + for target in linux/amd64 linux/arm64 darwin/arm64; do + goos="${target%/*}" + goarch="${target#*/}" + bin="staging/wavehouse-${goos}-${goarch}/wavehouse" + if [ ! -f "$bin" ]; then + echo "::error::missing build artifact for ${target} (${bin})" + exit 1 + fi + # actions/upload-artifact zips its input and that zip carries no + # unix mode bits, so the executable bit does NOT survive the round + # trip. Restore it before anything tars or COPYs the file. + chmod +x "$bin" + + stage="$(mktemp -d)" + cp "$bin" "$stage/wavehouse" + # NOTICE is not optional: the repo is Apache-2.0 and §4(d) makes + # every redistributor of these archives inherit an attribution + # obligation they cannot satisfy from a tarball carrying only + # LICENSE. CHANGELOG.md is deliberately absent — 300+ KB of + # development history, one click away on GitHub, and the release + # page already renders the notes. + cp LICENSE NOTICE README.md "$stage/" + tar -czf "dist/wavehouse_${goos}_${goarch}.tar.gz" \ + -C "$stage" wavehouse LICENSE NOTICE README.md + rm -rf "$stage" + + # Dockerfile.goreleaser COPYs `$TARGETPLATFORM/wavehouse`. + if [ "$goos" = linux ]; then + mkdir -p "ctx/linux/${goarch}" + cp "$bin" "ctx/linux/${goarch}/wavehouse" + fi + done + + # Base names only, two-space separator — the format GoReleaser wrote, + # and what `actions/attest-build-provenance` and `sha256sum -c` both + # expect. + (cd dist && sha256sum wavehouse_*.tar.gz > checksums.txt) + echo "--- dist/" + ls -l dist + echo "--- checksums.txt" + cat dist/checksums.txt + + # The GitHub-hosted runner's default docker driver can't build # multi-platform images, so create a docker-container builder. No QEMU - # needed — Dockerfile.goreleaser stages are $BUILDPLATFORM-pinned / - # COPY-only by design. Mirrors publish-dev.yml. + # needed — Dockerfile.goreleaser's stages are $BUILDPLATFORM-pinned or + # COPY-only by design, so nothing foreign-arch ever executes. - name: Set up Docker Buildx uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0 @@ -77,36 +233,90 @@ jobs: username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} - # Resolve the moving GHCR tag BEFORE the build, so a malformed tag - # fails here rather than after an 8-target cross-compile and a - # multi-arch image push. The script rejects anything that isn't - # semver-shaped, so a typo can't fall through to `latest`. - - name: Resolve release channel + # Before the GitHub Release, deliberately: a failed image build then + # leaves nothing public to reconcile, and the job is re-runnable. The + # reverse order would publish a release announcing an image that does + # not exist. + # + # Every build gets an immutable reference plus one moving pointer: + # :{tag} plus the channel release-channel.sh derived — :latest for a + # stable tag, :alpha/:beta/:rc/:next for a prerelease. That indirection + # is why `v1.3.0-rc.1` can't take :latest away from a shipped `v1.2.0`. + # + # The labels mirror what GoReleaser's dockers_v2.labels injected; the + # static floor for all of them lives in the Dockerfile itself. + - name: Build and push the multi-arch image shell: bash + env: + IMAGE: ghcr.io/wave-rf/wavehouse run: | set -euo pipefail - channel="$(scripts/ci/release-channel.sh "$GITHUB_REF_NAME")" - echo "WAVEHOUSE_CHANNEL=${channel}" >> "$GITHUB_ENV" - echo "::notice::${GITHUB_REF_NAME} publishes ghcr.io/wave-rf/wavehouse:${GITHUB_REF_NAME} + :${channel}" + version="${GITHUB_REF_NAME#v}" + created="$(date -u +%Y-%m-%dT%H:%M:%SZ)" + docker buildx build \ + --platform linux/amd64,linux/arm64 \ + --file deployments/Dockerfile.goreleaser \ + --tag "${IMAGE}:${GITHUB_REF_NAME}" \ + --tag "${IMAGE}:${WAVEHOUSE_CHANNEL}" \ + --label "org.opencontainers.image.title=WaveHouse" \ + --label "org.opencontainers.image.description=Schema-aware real-time API gateway for ClickHouse" \ + --label "org.opencontainers.image.url=${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}" \ + --label "org.opencontainers.image.source=${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}" \ + --label "org.opencontainers.image.version=${version}" \ + --label "org.opencontainers.image.revision=${GITHUB_SHA}" \ + --label "org.opencontainers.image.created=${created}" \ + --push \ + ctx - - name: Run GoReleaser - uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 - with: - # Pin to the v2 line so a v3 release breaks loudly instead - # of silently changing release behavior. - version: "~> v2" - args: release --clean + # GoReleaser's `release.mode` defaulted to `keep-existing`, which is what + # made "publish from the Releases UI, let the tag fire this workflow" + # work without clobbering a hand-written body. Reproduce that: if the + # release already exists (UI-created, or this is a re-run), upload the + # assets and leave the notes alone. + # + # `--notes-start-tag` is not optional. GitHub's own generator picks the + # previous release itself, and this repo publishes `clients/ts/v*` + # releases too — so without it a server release's notes can be diffed + # against an SDK release. `git describe --match 'v*'` is the same rule + # .goreleaser.yaml's `git.ignore_tags` applied. + - name: Create or update the GitHub Release + shell: bash env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - # Second dockers_v2 tag in .goreleaser.yaml. Stable → latest; - # prerelease → alpha/beta/rc/next, so an rc can't displace the - # :latest a shipped stable release owns. - WAVEHOUSE_CHANNEL: ${{ env.WAVEHOUSE_CHANNEL }} + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set -euo pipefail + tag="$GITHUB_REF_NAME" + + if gh release view "$tag" >/dev/null 2>&1; then + echo "::notice::release ${tag} already exists — uploading assets, leaving its notes untouched" + gh release upload "$tag" dist/wavehouse_*.tar.gz dist/checksums.txt --clobber + exit 0 + fi + + args=(--verify-tag --title "$tag" --generate-notes) + + # `latest` is release-channel.sh's answer for a stable tag and only + # for a stable tag, so this is the same judgement GoReleaser's + # `prerelease: auto` made — and GitHub's "Latest release" badge keys + # off exactly this flag. + if [ "$WAVEHOUSE_CHANNEL" != latest ]; then + args+=(--prerelease) + fi + + if previous="$(git describe --tags --abbrev=0 --match 'v*' "${tag}^" 2>/dev/null)"; then + echo "::notice::release notes for ${tag} generated since ${previous}" + args+=(--notes-start-tag "$previous") + else + echo "::notice::no earlier v* tag is reachable from ${tag} — notes will cover the full history" + fi + + gh release create "$tag" "${args[@]}" \ + dist/wavehouse_*.tar.gz dist/checksums.txt # Build-provenance attestations — free for public repos via Sigstore's # public-good infra. Binaries: one attestation over every artifact listed - # in goreleaser's checksums.txt. Image: attest the multi-arch - # manifest-list digest and store the attestation alongside it in GHCR. + # in checksums.txt. Image: attest the multi-arch manifest-list digest and + # store the attestation alongside it in GHCR. - name: Attest binary provenance uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2 with: @@ -132,10 +342,10 @@ jobs: # Prove the attestations we just wrote actually verify against the # artifacts people will download — the same `gh attestation verify` - # command, flags included, that the install docs tell users to run. Attesting and verifying - # are different code paths (subject digests, the checksums-file - # expansion, the registry round-trip), so a release that publishes - # unverifiable provenance should go red rather than look green. + # command, flags included, that the install docs tell users to run. + # Attesting and verifying are different code paths (subject digests, the + # checksums-file expansion, the registry round-trip), so a release that + # publishes unverifiable provenance should go red rather than look green. # Runs last: the release is already public by this point, so this is a # loud alarm, not a gate. - name: Verify published provenance @@ -152,7 +362,7 @@ jobs: # attestation. Verify them all rather than a representative one — # a per-GOOS/GOARCH gap is exactly what this would miss. shopt -s nullglob - archives=(dist/*.tar.gz dist/*.zip) + archives=(dist/*.tar.gz) if [ ${#archives[@]} -eq 0 ]; then echo "::error::no release archives in dist/ to verify" exit 1 diff --git a/.goreleaser.yaml b/.goreleaser.yaml index d809ce7d..c18851f2 100644 --- a/.goreleaser.yaml +++ b/.goreleaser.yaml @@ -1,21 +1,61 @@ version: 2 +# ───────────────────────────────────────────────────────────────────────────── +# GoReleaser is now ONLY the compiler. It is invoked once per native runner as +# `goreleaser build --single-target`, and everything downstream of the binary — +# archives, checksums, the GitHub Release, the multi-arch GHCR image — is done +# by .github/workflows/release.yml. This file therefore contains `builds:` and +# nothing else. +# +# Why, measured rather than assumed (2026-09-16, this tree at the cgo switch): +# +# cgo means each target needs a C toolchain that can target that OS/ABI, and +# for darwin that means real Apple SDK headers. In golang:1.27-bookworm +# (linux/amd64 — the shape of `ubuntu-latest`): +# +# linux/amd64, ambient gcc 12.2 OK 44,931,216 B max GLIBC_2.34 +# linux/arm64, apt gcc-aarch64-linux-gnu OK 41,400,592 B max GLIBC_2.34 +# linux/arm64, zig cc aarch64-linux-gnu.2.17 OK 51,094,216 B max GLIBC_2.17, +# but NEEDED libresolv/libdl/ +# libpthread and `-s` ignored +# darwin/arm64, zig cc -target aarch64-macos FAILS, with AND without +# `-tags netgo,osusergo`: +# +# # github.com/prometheus/client_golang/prometheus +# process_collector_mem_cgo_darwin.c:18:10: fatal error: +# 'mach/mach_vm.h' file not found +# +# That is a *compile* failure inside a dependency's own C file, not the +# `-lresolv` *link* failure recorded earlier — netgo/osusergo cannot help, +# because the build never reaches the linker. client_golang's darwin process +# collector switches on `cgo` with no opt-out tag, and zig ships no Mach +# headers. Supplying them needs an Apple SDK, which cannot be fetched on a +# GitHub-hosted Linux runner. So: no darwin cross-compile, at all. +# +# The two GoReleaser features that solve this natively — split/merge +# (`goreleaser continue --merge`) and `builder: prebuilt` — are Pro-only, and +# `goreleaser release --skip=` accepts no `build` value in OSS (checked +# against v2.18.1, whose accepted set is announce, archive, aur, aur-source, +# before, chocolatey, docker, flatpak, homebrew, iru, ko, makeself, mcp, +# nfpm, nix, notarize, publish, sbom, scoop, sign, snapcraft, srpm, validate, +# winget). There is therefore no OSS way to have GoReleaser assemble a +# release from binaries built elsewhere — hence the split above. +# ───────────────────────────────────────────────────────────────────────────── + # GoReleaser defaults ProjectName to the repo directory name — "WaveHouse" — # which is the only place the product's display casing leaked into an artifact # identifier: archives built as `WaveHouse_linux_amd64.tar.gz` while the binary # inside, the GHCR image, and the npm package are all lowercase `wavehouse`. # Asset URLs are permanent once a release is published, so this is pinned -# rather than inherited. +# rather than inherited. release.yml names its archives from this same value. project_name: wavehouse -# GoReleaser's tag detection — which tag it is releasing, and which one the -# release notes are diffed from — walks git history without caring which tag -# family it lands on. This repo has several: `v*` for the server and -# `clients//v*` for each client SDK. Without this, cutting `v0.1.0` after -# an SDK release describes the server release as "everything since -# clients/ts/v0.1.0" — verified: `previous=clients/ts/v0.1.0 current=v0.1.0`. -# Ignoring the client families makes the built-in detection correct, which is -# why no workflow needs to pass GORELEASER_PREVIOUS_TAG. +# GoReleaser's tag detection — which tag it is building from — walks git +# history without caring which tag family it lands on. This repo has several: +# `v*` for the server and `clients//v*` for each client SDK. Without +# this, a snapshot build cut after an SDK release stamps itself from +# `clients/ts/v0.1.0`. release.yml applies the same rule to the release notes +# by deriving `--notes-start-tag` from `git describe --match 'v*'`. git: ignore_tags: - "clients/*" @@ -24,114 +64,36 @@ builds: - id: wavehouse main: ./cmd/wavehouse binary: wavehouse + # cgo is unconditional: the chtypes binding dlopens the per-version + # artifact, so CGO_ENABLED=1 on every target (dlfcn only — no C library + # linked, no header). The platform set follows from that artifact, not + # from Go: Windows and FreeBSD are dropped + # (chtypes publishes no artifact for either) along with darwin/amd64 (no + # artifact either — only darwin-arm64, linux-amd64, linux-arm64 exist at + # https://artifacts.wavehouse.dev/artifacts/index.json). goos x goarch is + # a cross product, hence the explicit `ignore` to drop darwin/amd64 + # rather than a matrix that only ever had one cell per OS. env: - - CGO_ENABLED=0 - goos: [linux, darwin, windows, freebsd] + - CGO_ENABLED=1 + # `--single-target` builds the runner's own GOOS/GOARCH *regardless of what + # is set here*, so on the release path these three keys are a declaration + # rather than a driver: they are the supported set, and the `build` + # matrices in release.yml / publish-dev.yml / goreleaser-validate.yml must + # mirror them. Keep them accurate — `goreleaser check` validates this + # file's schema, nothing validates that a workflow matrix agrees with it, + # and a plain `goreleaser build` with no flag would still try (and fail) to + # cross-compile all three. + # + # There are deliberately NO `overrides:` with CC/CXX any more: every target + # is compiled by a native toolchain on its own runner, so the ambient `cc` + # is always the right one. + goos: [linux, darwin] goarch: [amd64, arm64] + ignore: + - goos: darwin + goarch: amd64 ldflags: - -s -w - -X main.Version={{.Version}} - -X main.BuildTime={{.Date}} - -X main.GitCommit={{.ShortCommit}} - -checksum: - name_template: "checksums.txt" - -archives: - - name_template: "{{ .ProjectName }}_{{ .Os }}_{{ .Arch }}" - # Windows gets .zip; everything else keeps GoReleaser's tar.gz default. - # Windows has shipped bsdtar since 10 1803, so a .tar.gz is *openable* — - # but not from Explorer, which still only double-clicks into .zip. The - # convention costs nothing and this is the first release whose asset - # names people will script against. - format_overrides: - - goos: windows - formats: [zip] - # Narrowed from GoReleaser's default, which also globs CHANGELOG* — 324 KB - # of development history in every one of the eight archives, for a file - # that is a click away on GitHub and whose contents are the release notes - # the download page already shows. - # NOTICE is not optional: the repo is Apache-2.0 and §4(d) makes every - # redistributor of these archives inherit an attribution obligation they - # cannot satisfy from a tarball carrying only LICENSE. GoReleaser's default - # glob never included it either, so this hunk is the moment to fix it — - # archive contents are effectively permanent once published. - files: - - LICENSE* - - NOTICE* - - README* - -# Suppress the GitHub Release on dev pushes. publish-dev.yml sets -# WAVEHOUSE_DEV=1 + a synthetic tag so it can reuse this same pipeline; -# we don't want a release entry on the repo for every push to main. -# Real tag releases (release.yml) leave WAVEHOUSE_DEV unset → defaults -# to "0" → the field evaluates to "false" → release runs normally. -# `envOrDefault` is required because goreleaser templates fail strict -# on missing env vars; bare `.Env.WAVEHOUSE_DEV` would error on the -# release path. -release: - disable: '{{ eq (envOrDefault "WAVEHOUSE_DEV" "0") "1" }}' - # `auto` marks the GitHub Release as a pre-release whenever the tag - # carries a prerelease identifier (v0.1.0-alpha.1 → yes; v0.1.0 → no). - # GoReleaser's default is a flat `false`, which would have published the - # first alpha as a full stable release — and GitHub's "Latest release" - # badge keys off exactly this field. - prerelease: auto - -dockers_v2: - - dockerfile: deployments/Dockerfile.goreleaser - ids: - - wavehouse - images: - - "ghcr.io/wave-rf/wavehouse" - # Every build gets an immutable reference plus one moving pointer. - # - # immutable release: :{{ .Tag }} (e.g. :v1.2.3) - # dev: :dev- (matches cleanup-ghcr.yml's - # `^dev-[0-9a-f]+$` regex) - # moving release: the channel release.yml derived from the tag via - # scripts/ci/release-channel.sh — :latest for a - # stable tag, :alpha/:beta/:rc/:next for a - # prerelease - # dev: :dev (rolling, follows main) - # - # The channel indirection is why `v1.3.0-rc.1` can't take :latest away - # from a shipped `v1.2.0`. The dev branch stays an explicit literal - # rather than leaning on WAVEHOUSE_CHANNEL's default: a push to main that - # forgot to set the env var must never be able to publish :latest. - tags: - - '{{ if eq (envOrDefault "WAVEHOUSE_DEV" "0") "1" }}dev-{{ .FullCommit }}{{ else }}{{ .Tag }}{{ end }}' - - '{{ if eq (envOrDefault "WAVEHOUSE_DEV" "0") "1" }}dev{{ else }}{{ envOrDefault "WAVEHOUSE_CHANNEL" "latest" }}{{ end }}' - platforms: - - linux/amd64 - - linux/arm64 - labels: - "org.opencontainers.image.title": '{{ if eq (envOrDefault "WAVEHOUSE_DEV" "0") "1" }}WaveHouse (dev){{ else }}WaveHouse{{ end }}' - "org.opencontainers.image.description": '{{ if eq (envOrDefault "WAVEHOUSE_DEV" "0") "1" }}Schema-aware real-time API gateway for ClickHouse — rolling dev build from main{{ else }}Schema-aware real-time API gateway for ClickHouse{{ end }}' - "org.opencontainers.image.url": "https://github.com/Wave-RF/WaveHouse" - "org.opencontainers.image.source": "{{ .GitURL }}" - "org.opencontainers.image.version": "{{ .Version }}" - "org.opencontainers.image.revision": "{{ .FullCommit }}" - "org.opencontainers.image.created": "{{ .Date }}" - -changelog: - # The changelog pipe is NOT skipped by `release.disable` — only by - # `--snapshot` or this key (verified: a run with `release.disable: true` - # still logs "generating changelog"). Left on, `github-native` would make - # publish-dev.yml POST /releases/generate-notes on every push to main — an - # endpoint needing `contents: write`, which that workflow deliberately does - # not grant — to build a body that is then discarded, for a synthetic tag - # that exists only on the runner. goreleaser-validate.yml cannot catch it - # either, since `build --snapshot` skips this pipe entirely. - disable: '{{ eq (envOrDefault "WAVEHOUSE_DEV" "0") "1" }}' - # Delegate the release body to GitHub's own "generate release notes" — the - # grouped, linked, per-PR list you get from the Releases UI button, with - # authors and a New Contributors section. `use: git` built the body from raw - # commit subjects instead: nearly the same content (main is squash-merged, so - # each subject IS a PR title) but no links, no authors, and no grouping. - # - # Categorization and exclusions move to .github/release.yml with this. The - # `sort` and `filters` keys that used to live here are gone rather than left - # in place: github-native renders the body on GitHub's side, so GoReleaser - # never sees the commits and both would be silently dead config. - use: github-native diff --git a/Makefile b/Makefile index 85dc1dda..4f9f4045 100644 --- a/Makefile +++ b/Makefile @@ -160,7 +160,19 @@ GSA := GOEXPERIMENT=jsonv2 go tool gsa # Externally-installed tools — version is encoded in the path so bumping the # version invalidates the file rule and triggers a reinstall. -GOLANGCI_LINT_VERSION := v2.11.4 +# +# v2.11.4 panics in its type-checker on every package once go.mod says +# `go 1.27`. v2.13.0 is the OLDEST release whose +# changelog claims go1.27 support ("go1.27 support (#6642)"), but it panics +# too — a DIFFERENT bug: `nilness`/`honnef.co/go/tools@v0.8.0-rc.1` crashes +# analyzing a third-party dependency's source (measured: `internal error: +# unhandled builtin recover`, package "sentry", i.e. getsentry/sentry-go). +# v2.13.1 bumps that dependency past its release candidate to the real +# 0.8.0 and the panic is gone; v2.13.2 (bumping it again, to 0.8.1) is the +# newest confirmed-clean release at time of writing — pinned here rather +# than v2.13.1 since nothing points at 2.13.1 specifically being the fix, +# only that 2.13.0 is broken and 2.13.2 is verified clean in this tree. +GOLANGCI_LINT_VERSION := v2.13.2 GOLANGCI_LINT := $(LOCAL_BIN)/golangci-lint-$(GOLANGCI_LINT_VERSION) # air is the hot-reload runner used by `make dev`. We install it to .bin/ @@ -171,7 +183,7 @@ AIR_VERSION := v1.65.1 AIR := $(LOCAL_BIN)/air-$(AIR_VERSION) # misspell: curated common-typo corrector + US/UK locale enforcer. Installed -# standalone to .bin/ (pure Go, `go install` — same pattern as air) so it can +# standalone to .bin/ via `go install`, same pattern as air, so it can # lint Markdown/MDX prose. DISTINCT from the misspell analyzer bundled inside # golangci-lint, which only inspects Go source; same maintained fork # (github.com/golangci/misspell), two entry points. Drives `make lint-prose`. @@ -912,20 +924,6 @@ release-sdk-go: ## Tag a Go SDK release — go get (VERSION=X.Y.Z) # `verify` or `ci` — `deadcode` has false positives on reflection / HTTP # routers, and the size/dep tools are too slow for a pre-push gate. -# audit-cgo: WaveHouse builds with CGO_ENABLED=0. Listed packages have pure-Go -# fallbacks today, but a new dep could quietly break that constraint — this -# audit surfaces every transitively-reachable package with C files so the drift -# is visible before a release-time cross-compile breaks. -.PHONY: audit-cgo -audit-cgo: ## Audit dependency tree for CGO files (informational) - @echo "$(CYAN)==> Scanning dependency tree for packages with C files...$(RESET)" - @printf " WaveHouse builds with %sCGO_ENABLED=0%s — listed packages have pure-Go fallbacks\n" "$(YELLOW)" "$(RESET)" - @echo " and their C code is never compiled. This audit catches new CGO deps." - @echo - @CGO_ENABLED=1 go list -deps -f '{{if .CgoFiles}} ⚠ {{.ImportPath}} ({{len .CgoFiles}} C files){{end}}' ./cmd/... - @echo - @echo "$(GREEN)==> CGO audit complete$(RESET)" - # deadcode: whole-program reachability analysis, complementary to # golangci-lint's `unused` (which is locally scoped). False positives are # common for HTTP routers, reflection-based dispatch, and init() registration @@ -952,9 +950,9 @@ dep-cut: ## Top cuttable dependencies by transitive weight (LIMIT=N to override) @LIMIT='$(LIMIT)' scripts/dep-cut.sh # binary-analysis: one command for "what's in my binary, and what's wrong with -# it." Runs in dep-order: build → size → audit-cgo → deadcode. +# it." Runs in dep-order: build → size → deadcode. .PHONY: binary-analysis -binary-analysis: size audit-cgo deadcode ## Combined: size + audit-cgo + deadcode +binary-analysis: size deadcode ## Combined: size + deadcode @echo @echo "$(GREEN)==> Binary analysis complete$(RESET)" @printf " Cuttable dependencies: %smake dep-cut%s\n" "$(CYAN)" "$(RESET)" @@ -1046,7 +1044,7 @@ $(GOLANGCI_LINT): @mv $(LOCAL_BIN)/golangci-lint $@ @echo "$(GREEN)==> Installed: $@$(RESET)" -# air installs cleanly via `go install` (pure Go, no shell-piping). GOBIN +# air installs cleanly via `go install` (no shell-piping). GOBIN # pins the install location to our .bin/ rather than the user's $GOPATH/bin. $(AIR): @echo "$(YELLOW)==> Installing air $(AIR_VERSION) for $(OS)_$(ARCH)...$(RESET)" @@ -1055,7 +1053,7 @@ $(AIR): @mv $(LOCAL_BIN)/air $@ @echo "$(GREEN)==> Installed: $@$(RESET)" -# misspell installs cleanly via `go install` (pure Go), GOBIN-pinned to .bin/ +# misspell installs cleanly via `go install`, GOBIN-pinned to .bin/ # like air. cmd/misspell is the CLI entry point of the golangci fork — the same # codebase golangci-lint vendors as a library for its Go-only misspell linter. $(MISSPELL): diff --git a/chtypes.lock b/chtypes.lock new file mode 100644 index 00000000..f65d419f --- /dev/null +++ b/chtypes.lock @@ -0,0 +1,17 @@ +{ + "schema": 1, + "artifacts": { + "darwin-arm64/26.6": { + "file": "chtypes-26.6.8.7-stable-darwin-arm64-b1790767905.tar.gz", + "sha256": "11df5c31e638990c595670d6b92e9e8382a7216fcba433ae0f6fbe5a325ab8f6" + }, + "linux-amd64/26.6": { + "file": "chtypes-26.6.8.7-stable-linux-amd64-b1790767905.tar.gz", + "sha256": "2fcbe729ae110507af22d35a19c290c97c1fae9f036aa2183841331b8ec9770e" + }, + "linux-arm64/26.6": { + "file": "chtypes-26.6.8.7-stable-linux-arm64-b1790767905.tar.gz", + "sha256": "c85effdb5512aac0e012304249b09c86b516f53b401613f251222267453351d3" + } + } +} diff --git a/deployments/Dockerfile b/deployments/Dockerfile index bb74ecd7..d4b5c6ae 100644 --- a/deployments/Dockerfile +++ b/deployments/Dockerfile @@ -1,15 +1,20 @@ # syntax=docker/dockerfile:1 # Stage 1: Build -FROM golang:1.26-alpine AS builder +# +# bookworm (glibc), not alpine (musl): the chtypes SDK's dlopen path needs +# cgo (dlfcn), and the image the SDK builds its own artifacts against — plus +# the artifact's own glibc floor (2.17, 2.29 for the 24.8/25.3 lines) — is +# glibc, not musl. The non-alpine golang image bundles gcc out of the box, +# unlike alpine (which needs an explicit apk add build-base for cgo). +FROM golang:1.27-bookworm AS builder WORKDIR /src # Pass build tags as an argument (defaults to empty) ARG BUILD_TAGS="" -# Download dependencies first into the module cache. The `golang:*-alpine` -# image sets GOPATH=/go, so `go mod download` writes to /go/pkg/mod — that's -# the cache mount target. Splitting this into its own layer means dependency -# changes (go.mod / go.sum) don't bust the source-code COPY layer's cache. +# Download dependencies first into the module cache. Splitting this into its +# own layer means dependency changes (go.mod / go.sum) don't bust the +# source-code COPY layer's cache. COPY go.mod go.sum ./ RUN --mount=type=cache,target=/go/pkg/mod go mod download @@ -20,9 +25,25 @@ COPY . . # tree (read by `go build` for imports) and /root/.cache/go-build for the # compiled-package cache (Go's content-addressed build cache). Without both, # every `docker build` is a cold compile. +# +# CGO_ENABLED=1 is unconditional (see the bookworm comment above). +RUN --mount=type=cache,target=/go/pkg/mod \ + --mount=type=cache,target=/root/.cache/go-build \ + CGO_ENABLED=1 go build -tags="${BUILD_TAGS}" -ldflags="-s -w" -o /bin/wavehouse ./cmd/wavehouse + +# Bake the pinned chtypes artifact(s) into the image so the container has no +# runtime network dependency and no first-request download stall. chtypes.lock +# is this repo's pin (schema-1, exact file + sha256 per platform/line); +# scripts/fetch-chtypes.sh wraps the SDK's own CLI with --frozen, which +# refuses anything the lock doesn't name. `go run @` +# resolves and builds the SDK's CLI straight from its module proxy — it +# needs no entry in THIS repo's go.mod/go.sum, so this works even before +# `go mod tidy` has picked up the chtypes/go dependency. +COPY chtypes.lock ./ RUN --mount=type=cache,target=/go/pkg/mod \ --mount=type=cache,target=/root/.cache/go-build \ - CGO_ENABLED=0 go build -tags="${BUILD_TAGS}" -ldflags="-s -w" -o /bin/wavehouse ./cmd/wavehouse + mkdir -p /opt/chtypes/artifacts && \ + scripts/fetch-chtypes.sh --dest /opt/chtypes/artifacts # Create the parent state and settings directories owned by the nonroot # user (UID 65532 in distroless). The binary creates `nats/` and `pebble/` @@ -42,13 +63,24 @@ RUN --mount=type=cache,target=/go/pkg/mod \ RUN mkdir -p /app/data /app/settings && chown -R 65532:65532 /app # Stage 2: Final minimal image -FROM gcr.io/distroless/static-debian12 +# +# distroless/cc, not distroless/static: chtypes' dlopen path needs glibc's +# dynamic loader plus libstdc++ (the artifact is a C++-built .so) — the +# "cc" variant bundles exactly that (glibc, libgcc1, libstdc++6) and nothing +# else, still with no shell or package manager. +FROM gcr.io/distroless/cc-debian12 WORKDIR /app USER nonroot:nonroot ENV WH_SETTINGS_DIR=/app/settings +# The SDK's own registry search path reads this directly — no +# WaveHouse config change is needed to find what the image already baked +# in. WH_CHTYPES_REGISTRY (internal/config) is for an operator who wants to +# point at a different, bind-mounted registry directory instead. +ENV CHTYPES_REGISTRY=/opt/chtypes/artifacts COPY --from=builder --chown=nonroot:nonroot /app /app +COPY --from=builder --chown=nonroot:nonroot /opt/chtypes /opt/chtypes COPY --from=builder --chown=nonroot:nonroot /bin/wavehouse /app/wavehouse # OCI image metadata. Static labels live here; per-build dynamic ones diff --git a/deployments/Dockerfile.goreleaser b/deployments/Dockerfile.goreleaser index 8d22201c..b2b54b77 100644 --- a/deployments/Dockerfile.goreleaser +++ b/deployments/Dockerfile.goreleaser @@ -1,8 +1,19 @@ +# The prebuilt-binary image. Despite the name, GoReleaser no longer drives +# this file: cgo made cross-compiling impossible for darwin, so the release +# pipeline builds each binary on its own native runner and +# .github/workflows/release.yml (and publish-dev.yml, and the no-push +# validation in goreleaser-validate.yml) invokes `docker buildx build` on it +# directly. The build CONTRACT is unchanged and those workflows reproduce it +# exactly — a context holding `//wavehouse` per target plus +# chtypes.lock and scripts/{fetch-chtypes,_colors}.sh — which is why this file +# needed no edit beyond these comments. The name is kept so published image +# history, `docker history` output and every doc reference stay valid. +# # Tiny stage just to mkdir + chown the state directories. Distroless has no # shell, so the runtime image can't run RUN commands itself; we materialise -# the layout here and COPY it into the final image. Keeps the goreleaser -# variant in lockstep with deployments/Dockerfile (source-build) so the two -# images expose the same /app filesystem layout to operators. +# the layout here and COPY it into the final image. Keeps this variant in +# lockstep with deployments/Dockerfile (source-build) so the two images expose +# the same /app filesystem layout to operators. # # Only the parents are pre-created — the binary mkdirs `nats/` and `pebble/` # subdirs itself when NATS/Pebble open their stores. Named-volume copy-up @@ -21,28 +32,62 @@ FROM --platform=$BUILDPLATFORM alpine:3 AS layout # RUN in deployments/Dockerfile; keep the two in lockstep. RUN mkdir -p /app/data /app/settings && chown -R 65532:65532 /app -FROM gcr.io/distroless/static-debian12 +# Fetches the pinned chtypes artifact for THIS image's target platform (not +# necessarily the builder's — buildx may emulate or cross-fetch). Built +# natively on $BUILDPLATFORM like `layout` above: no compiled Go code +# crosses an arch boundary here, just an HTTPS download + sha256 check via +# the SDK CLI's own --platform flag, so QEMU emulation would buy nothing. +# Needs chtypes.lock + scripts/fetch-chtypes.sh + scripts/_colors.sh at their +# normal repo-relative paths. The build context is assembled by the workflow +# rather than being the repo root, so those three are copied into it +# explicitly — see the "Assemble the image build context" step in +# .github/workflows/release.yml, publish-dev.yml and goreleaser-validate.yml. +# +# This stage is also the pipeline's chtypes.lock check: `--frozen` refuses any +# artifact the lock does not name, so an upstream republish of a pinned line +# fails the build here rather than silently baking a different library. +FROM --platform=$BUILDPLATFORM golang:1.27-bookworm AS chtypes-fetch +ARG TARGETPLATFORM +WORKDIR /src +COPY chtypes.lock ./ +COPY scripts/fetch-chtypes.sh scripts/_colors.sh ./scripts/ +RUN --mount=type=cache,target=/go/pkg/mod \ + --mount=type=cache,target=/root/.cache/go-build \ + plat="$(echo "$TARGETPLATFORM" | tr / -)"; \ + mkdir -p /opt/chtypes/artifacts && \ + scripts/fetch-chtypes.sh --platform "$plat" --dest /opt/chtypes/artifacts + +FROM gcr.io/distroless/cc-debian12 -# GoReleaser automatically injects this based on the platform it's building for +# buildx injects this per target platform of a `--platform a,b` build ARG TARGETPLATFORM WORKDIR /app USER nonroot:nonroot ENV WH_SETTINGS_DIR=/app/settings +# The SDK's own registry search path reads this directly — no +# WaveHouse config change is needed to find what the image already baked +# in. WH_CHTYPES_REGISTRY (internal/config) is for an operator who wants to +# point at a different, bind-mounted registry directory instead. +ENV CHTYPES_REGISTRY=/opt/chtypes/artifacts # Pre-created state directories owned by the nonroot user (UID 65532) so # the binary can mkdir under /app/data without a volume mount, and so # bind-mounting empty volumes at /app/data inherits correct ownership. COPY --from=layout --chown=nonroot:nonroot /app /app +COPY --from=chtypes-fetch --chown=nonroot:nonroot /opt/chtypes /opt/chtypes -# Copy the binaries from the platform-specific subdirectories GoReleaser creates +# Copy the binary from its platform-specific subdirectory of the context — +# `linux/amd64/wavehouse`, `linux/arm64/wavehouse`. The workflow stages each +# native runner's binary there with mode 0755 (actions/upload-artifact's zip +# carries no unix mode bits, so the executable bit has to be restored). COPY --chown=nonroot:nonroot $TARGETPLATFORM/wavehouse /app/wavehouse -# OCI image metadata. Static labels here; goreleaser injects per-build -# ones (revision, version, created, image.title with the version) via -# `dockers_v2.labels` in `.goreleaser.yaml`, which take precedence on -# release builds. The static set below is the floor — guaranteed even -# if a label is missing from the goreleaser config. +# OCI image metadata. Static labels here; the publishing workflows pass the +# per-build ones (revision, version, created, and a dev-suffixed title and +# description on the :dev channel) as `docker buildx build --label`, which +# take precedence. The static set below is the floor — guaranteed even when a +# label is missing from the command line, e.g. on a local `docker build`. LABEL org.opencontainers.image.title="WaveHouse" LABEL org.opencontainers.image.description="Schema-aware real-time API gateway for ClickHouse" LABEL org.opencontainers.image.url="https://github.com/Wave-RF/WaveHouse" diff --git a/internal/app/discoveries.go b/internal/app/discoveries.go index f056b470..1f86079e 100644 --- a/internal/app/discoveries.go +++ b/internal/app/discoveries.go @@ -137,7 +137,7 @@ func (d *discoveries) adopt(id tenant.ID, reg *discovery.SchemaRegistry) { // start runs reg's loop: the boot retry until the first success, skipped // for a registry already loaded, then the periodic refresh. func (d *discoveries) start(id tenant.ID, reg *discovery.SchemaRegistry) *tenantDiscovery { - ctx, cancel := context.WithCancel(d.ctx) //nolint:gosec // G118: held on the tenantDiscovery, called by reconcile or close + ctx, cancel := context.WithCancel(d.ctx) td := &tenantDiscovery{id: id, registry: reg, cancel: cancel, done: make(chan struct{})} go func() { defer close(td.done) diff --git a/internal/mq/external.go b/internal/mq/external.go index 778cdac0..30fa0684 100644 --- a/internal/mq/external.go +++ b/internal/mq/external.go @@ -241,7 +241,7 @@ func NewNATS(ctx context.Context, cfg NATSConfig) (*ExternalNATS, error) { e.nc.Close() return nil, fmt.Errorf("register mq gauges: %w", err) } - e.stopping, e.stop = context.WithCancel(context.Background()) //nolint:gosec // G118: Close calls it + e.stopping, e.stop = context.WithCancel(context.Background()) go e.watch(orDefault(cfg.recheckEvery, topologyRecheck), orDefault(cfg.historyEvery, historyPoll)) return e, nil } @@ -969,7 +969,7 @@ func (e *ExternalNATS) ReplaySince(ctx context.Context, topic Topic, since time. if err := ctx.Err(); err != nil { return err } - batch, err := cons.Fetch(int(min(remaining, replayBatch)), jetstream.FetchMaxWait(replayPullWait)) //nolint:gosec // capped + batch, err := cons.Fetch(int(min(remaining, replayBatch)), jetstream.FetchMaxWait(replayPullWait)) if err != nil { return fmt.Errorf("replay fetch: %w", err) } @@ -1042,7 +1042,7 @@ func (e *ExternalNATS) Stats() (observability.MQStats, error) { s := e.nc.Stats() return observability.MQStats{ Connections: boolGauge(e.nc.IsConnected()), - InMsgs: int64(min(s.InMsgs, uint64(1<<62))), //nolint:gosec // capped + InMsgs: int64(min(s.InMsgs, uint64(1<<62))), }, nil } diff --git a/scripts/build.sh b/scripts/build.sh index f628f23b..2f4aa743 100755 --- a/scripts/build.sh +++ b/scripts/build.sh @@ -64,7 +64,10 @@ printf '%s==> Building %s...%s\n' "$CYAN" "$label" "$RESET" start=$(date +%s) # build_flags may be empty; use the ${arr[@]+"${arr[@]}"} idiom so the empty # expansion is silent under `set -u` on bash 3.2 (macOS default). -CGO_ENABLED=0 go build \ +# +# CGO_ENABLED=1 is unconditional: the chtypes SDK's dlopen path needs cgo for +# dlfcn (no C library to link against, no header). +CGO_ENABLED=1 go build \ -tags="${TAGS:-}" \ ${build_flags[@]+"${build_flags[@]}"} \ -ldflags="$ldflags" \ diff --git a/scripts/fetch-chtypes.sh b/scripts/fetch-chtypes.sh new file mode 100755 index 00000000..a9695674 --- /dev/null +++ b/scripts/fetch-chtypes.sh @@ -0,0 +1,98 @@ +#!/usr/bin/env bash +# Fetch the chtypes artifact(s) this repo pins in chtypes.lock, via the +# published SDK CLI. One wrapper, three callers: +# - deployments/Dockerfile / Dockerfile.goreleaser (bakes the artifact +# into the runtime image at build time) +# - .github/actions/setup-env (CI cache-miss fetch, before Go test jobs) +# - developers (`scripts/fetch-chtypes.sh` with no args pulls the host +# platform's build of the e2e harness's pinned line) +# +# Always --frozen --lock chtypes.lock: this repo's lock is the only source +# of truth for which exact build gets installed — never the rolling +# `artifacts` release (the SDK documents the lock as +# refreshed deliberately, never regenerated implicitly). +# A line not in the lock, or a lock/registry mismatch, is a hard failure. +# +# Usage: scripts/fetch-chtypes.sh [ ...] [--platform ] [--dest ] +# ClickHouse minor line(s) to fetch, e.g. 26.6. Defaults to +# every line this repo needs today (LOCK_LINES below). +# --platform os-arch pair to fetch for (default: host platform, chosen +# by the SDK's own HostPlatform()). Pass linux-amd64 / +# linux-arm64 to fetch a foreign platform's artifact — used +# when baking a Linux container image from a non-Linux host. +# --dest registry directory to fetch into (default: the SDK's own +# default per-platform cache dir — see `chtypes where`). +# +# Exit codes are the SDK CLI's own: 0 ok, 1 verification/lock mismatch, +# 2 usage, 3 source unreachable, 4 not published. +set -euo pipefail + +# shellcheck source=scripts/_colors.sh +. "$(dirname "$0")/_colors.sh" + +# The SDK's own package is cgo-gated, so building the CLI below without a C +# compiler on PATH fails with a pile of "undefined: Registry" / "undefined: +# minorOf" errors from files the build tags excluded — which reads like a +# broken dependency, not a missing toolchain. Say so instead. (Measured in a +# bare ubuntu:24.04; every CI runner and golang:*-bookworm already ship one.) +if [ "$(go env CGO_ENABLED 2>/dev/null || echo 0)" != "1" ]; then + printf '%sno C compiler: go env CGO_ENABLED is not 1%s\n' "${RED}" "${RESET}" >&2 + printf ' The chtypes SDK is cgo-only. Install a C toolchain (Debian/Ubuntu:\n' >&2 + printf ' apt-get install gcc; macOS: xcode-select --install) and retry.\n' >&2 + exit 2 +fi + +# Bump together with go.mod's `require github.com/wave-rf/chtypes/go` line. +CHTYPES_SDK_VERSION="v0.4.0" +CHTYPES_CLI="github.com/wave-rf/chtypes/go/cmd/chtypes@${CHTYPES_SDK_VERSION}" + +REPO_ROOT="$(cd "$(dirname "$0")/.." && pwd)" +LOCK_FILE="${REPO_ROOT}/chtypes.lock" + +# Lines every deployment of this repo needs today. The e2e harness +# (tests/integration/setup_test.go, scripts/orchestrator) and dev compose +# both pin ClickHouse 26.6.3.62; chtypes resolves by MINOR line (26.6), +# never nearest, so this is "26.6", not the exact patch. Widening +# this list is how a new line gets adopted: fetch it, add it here, commit +# the updated lock. +LOCK_LINES=(26.6) + +platform="" +dest="" +lines=() + +while [ $# -gt 0 ]; do + case "$1" in + --platform) + platform=${2:?--platform requires a value} + shift 2 + ;; + --dest) + dest=${2:?--dest requires a value} + shift 2 + ;; + -h | --help) + printf 'Usage: %s [ ...] [--platform ] [--dest ]\n' "$0" + exit 0 + ;; + -*) + printf '%sunknown flag: %s%s\n' "${RED}" "$1" "${RESET}" >&2 + exit 2 + ;; + *) + lines+=("$1") + shift + ;; + esac +done + +if [ ${#lines[@]} -eq 0 ]; then + lines=("${LOCK_LINES[@]}") +fi + +args=(fetch "${lines[@]}" --frozen --lock "$LOCK_FILE") +[ -n "$platform" ] && args+=(--platform "$platform") +[ -n "$dest" ] && args+=(--dest "$dest") + +printf '%s==> chtypes fetch --frozen%s %s (lock: %s)\n' "${CYAN}" "${RESET}" "${lines[*]}" "$LOCK_FILE" +exec go run "$CHTYPES_CLI" "${args[@]}" diff --git a/scripts/size.sh b/scripts/size.sh index 91b73363..518123c8 100755 --- a/scripts/size.sh +++ b/scripts/size.sh @@ -47,8 +47,10 @@ human() { # Heads-up on a single common point of confusion in the gsa output. # Section-name reference lives in docs/, not reprinted every run. printf '%s==> Reading the gsa output:%s\n' "$CYAN" "$RESET" -printf ' %s"CGO" rows are mostly Go reflection metadata, not C code%s — this build is CGO_ENABLED=0.\n' \ +printf ' %s"CGO" rows are mostly Go reflection metadata, not linked C code%s — chtypes needs cgo for\n' \ "$YELLOW" "$RESET" +printf ' dlfcn, but it dlopens its artifact at runtime rather than linking a C library, so it adds\n' +printf ' only a thin shim to this bucket.\n' printf ' Focus on large NAMED packages (vendor / std); treat CGO/Unknown rows as noise.\n\n' # ── Side-by-side debug / release comparison. From e58d17f490bc5a5e049dc289c001a2e7e3eac4ee Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:31:34 -0400 Subject: [PATCH 02/70] docs: describe the chtypes type layer on the multi-tenant text Ports the reference migration's documentation onto main's current multi-tenant, multi-role text. Ingest validation and row-level security run through one process-wide chtypes engine with a per-tenant table set, loaded only by api-role processes; a tenant with no artifact for its ClickHouse line, or a server time zone that differs from the zone that line was opened with, is refused on its own. Error bodies keep the string code class and add exception_code. /v1/query and pipes are rendered by ClickHouse. Platforms narrow to linux/amd64, linux/arm64 and darwin/arm64 on glibc. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- AGENTS.md | 23 +- CHANGELOG.md | 16 ++ README.md | 16 +- docs/src/content/docs/access-control.mdx | 51 ++-- docs/src/content/docs/api.md | 283 +++++++++---------- docs/src/content/docs/architecture.md | 113 +++++--- docs/src/content/docs/configuration.mdx | 3 +- docs/src/content/docs/deployment.md | 52 +++- docs/src/content/docs/development.md | 43 ++- docs/src/content/docs/getting-started.md | 6 +- docs/src/content/docs/index.mdx | 8 +- docs/src/content/docs/ingest-pipeline.md | 11 +- docs/src/content/docs/reverse-proxy.mdx | 4 +- docs/src/content/docs/sdk/queries.md | 4 +- docs/src/content/docs/sdk/reference.md | 4 +- docs/src/content/docs/sdk/streaming.md | 4 +- docs/src/content/docs/settings-directory.mdx | 4 +- docs/src/content/docs/why-wavehouse.md | 2 +- 18 files changed, 377 insertions(+), 270 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index b29e203e..c95c3c82 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -26,9 +26,9 @@ One binary: - **`cmd/wavehouse/`** — Standalone mode (all-in-one with embedded NATS, optional Pebble dedup): argv dispatch, the logger, `config.Load`, and the signal context; everything else is `internal/app`. The boot config's `roles` (`api`, `ingest`, `sweeper`; all by default) pick which components one process runs, so the same binary can be one Deployment per role -Twenty internal packages under `internal/` (plus `internal/testutil/` for shared test helpers): +Twenty-one internal packages under `internal/` (plus `internal/testutil/` for shared test helpers): -- **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers; `ch_errors.go` (`writeCHError`) is the one mapping from a failed ClickHouse query to status, `code` and `retryable` (`writeCHWriteError` for a write pipe: never retryable, no `Retry-After`, since the write may have run) +- **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers. On the ingest path, `content_type.go` is the Content-Type → format table (and the RFC 9110 resolution rules behind the `415`) and `ingest_framing.go` is the only code that reads body bytes itself — the array re-frame, the record count, and the positional dedupe-id read. `clickhouse_http.go` runs structured queries and pipes over each tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, so ClickHouse renders every row; `ch_errors.go` (`writeCHError`) is the one mapping from a failed ClickHouse query to status, `code` and `retryable` (`writeCHWriteError` for a write pipe: never retryable, no `Retry-After`, since the write may have run) - **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive` for the one setting folded over every tenant served, `gapWindows` handing the sweeper each tenant's own gap window (a rejected tenant's as its folder last had it, unbounded for one rejected since boot) and the `mq.max_bytes_gb` reconcile each served tenant's byte budget, and `defaultPolicy` for the one setting that still follows tenant `0`, a flat directory's ops-gate admin role; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, the same hook's `Hub.Prune` ends the open streams of a tenant no longer served, and `wireCache`'s hook drops, through `LocalCache.Prune`, the cache version index of a tenant no longer served ([#262](https://github.com/Wave-RF/WaveHouse/issues/262))), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `New` wires only what the process's `roles` need (discovery, dedupe, auth verifiers, the hub bridge and keepalive per API process; the ingest worker per ingest process; the sweeper under its lease through `elected`, on the embedded MQ only); a process without `api` serves `api.NewOpsRouter` — probes, `/version`, metrics, and the settings reload behind the operator key alone. `cmd/wavehouse` and `tests/integration` both boot through it - **`auth/`** — JWT auth middleware: HMAC **or** JWKS verification with `alg` pinned to the active verifier, role extraction from a configurable claim path; always runs, never rejects (bad token → empty role + stashed reason). One verifier per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9): `Authenticator` keys them by `tenant.ID` — the request store's `settings.Store.Tenant()`, through an injected `TenantSource`; `tenant.Default` on the tenant-exempt routes — built from each tenant's `auth` block by `Reconfigure`, dropped by `Prune` once the tenant stops being served (rejected or removed), released by `Close`; the secrets (`Config`) are boot-level and shared. A JWKS key set is fetched off the boot and reload paths: until one has been stored the verifier is pending and a token-bearing request gets `503` + `Retry-After` from `api.refuseUnverifiable` (`auth.ErrVerifierPending`), never a `default_role` evaluation; refresh is library-managed (Eric, 2026-09-22), response capped at 1 MiB; the operator key's admin role is the request tenant's - **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index), and `RedisCache`, the Redis-compatible shared backend (random version tokens under the tenant's hash tag, one-round-trip lookups, bypass on failure behind a circuit breaker, deferred invalidations retried; selected by `cache.backend: redis`, configured by the boot config's `cache.redis` block — [#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every key carries the tenant (in `RedisCache`, after the key prefix: `:{}:…` for a version token, `:q::…` for a value); in `LocalCache` and the version index it leads ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for the caller's query key and its singleflight, escaped whole as the lead field of the stored key `|.||…`, where each raw table and scope name is escaped by `keyenc` (a `Namespace` carries them raw, so no caller escapes); the index holds a version per tenant, per (tenant, table) and per (tenant, table, scope), keyed by raw name and bumped in place (one entry per live namespace however often it is bumped, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)) — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` drops the tenant's index so its next key gets a process-unique generation, orphaning its every cached result in one step, pipe results included (no insert reaches a pipe result until [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); `Lookup` returns a `Snapshot` of the versions it read, taken before the handler chooses any input a bump invalidates — the tenant's connection included — and `Set` files the fill under it, so a write landing mid-query, or a reload moving the tenant to another address or database after the request took its connection, orphans the fill ([#382](https://github.com/Wave-RF/WaveHouse/issues/382)), and every backend runs the conformance suite `internal/testutil/cachetest`; the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the whole cache of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) @@ -37,8 +37,8 @@ Twenty internal packages under `internal/` (plus `internal/testutil/` for shared - **`config/`** — YAML + env var config loading (cleanenv); strict on both sides (undeclared YAML key, unbound `WH_*` variable) and probes `data_dir` writability when a selected backend keeps state there (`NeedsDataDir`); `backends.go` holds each layer's `.backend` (the in-process value by default; `mq.backend` also takes `nats`, with its `mq.nats` sub-block of file-path-only credentials; `coord.backend` takes `nats`, whose `coord.nats` block names only the lease bucket and rides `mq.nats`'s connection (both blocks, their rules and warnings are `mq_nats.go`); `cache.backend` takes `redis`, whose sub-block is `cache_redis.go`; and `dedupe.backend` takes `dynamodb`, with its `dedupe.dynamodb` sub-block) and `Warnings`, the valid combinations boot logs at `WARN`; `config.go` holds `roles` (`Has(Role)`) and `instance_id`, and `Validate` refuses a role split the backends cannot serve (any split over the embedded MQ; `api` without `ingest`, or the reverse, over a local cache; `coord.backend=nats` without `mq.backend=nats`; `mq.backend=nats` with `coord.backend=local` in a process running `ingest`; a process running only `sweeper` under `mq.backend=nats`) — boot is the validator, there is no dry run - **`coord/`** — leases for work that must run in one process at a time (`Observer.Held` reads whether one is held without campaigning): `Coordinator.TryAcquire(ctx, name)` → a `Term` (fencing `Token`, strictly increasing per name; `Done`/`Err`, `ErrLost` on loss; `Resign`), `ErrHeld` while another holder's — or this coordinator's own — term is live; `RunElected` runs a loop only while holding its lease, resigning when the loop returns and campaigning again every `RetryPeriod`. `Local` is the in-process implementation (first taker wins, never expires; `Peer` is a second handle over the same table for tests); every implementation runs `coordtest.Conformance`. Imports only the standard library, so a distributed backend lives beside its connection: `coord.backend: nats` is `internal/mq/lease.go` (`ExternalNATS.Leases`), a key per lease in the operator's KV bucket, the KV revision as the fencing token, and expiry judged on the candidate's own clock (the same revision seen unchanged for 15s), never by a server TTL. `internal/app`'s `wireCoord` opens the one `coord.backend` selects and the sweeper runs through `RunElected` under the `sweeper` lease - **`dedupe/`** — `Deduplicator` interface (two-phase `Reserve`/`Commit`/`Release` over `Key{Table, ID}`; every backend passes the `dedupetest` conformance suite) → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant and table, pending claims in memory, committed ids stored with their expiry and deleted by an hourly background sweep along with the version-0 keys from before the table joined the key, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges) or `Dynamo` (one shared DynamoDB table, conditional `PutItem` claims; conformance-tested against dynamodb-local, selected by `dedupe.backend: dynamodb`; boot checks the table and never creates it outside dynamodb-local), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` or, gated on the table check (`Factory.Gated`), `Dynamo.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) -- **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`, and starts a tenant over on a fresh registry when a reload moves it to another address or database: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version, + `Validate()` for ingest payloads + `CanonicalizeTimestamps()` rewriting top-level `DateTime`/`DateTime64` column values to the canonical RFC 3339 UTC wire form pre-publish (Key Design Decision #19) -- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else — `row` is one positional `JSONCompactEachRow` line and `columns` names its slots, the table's **insertable** columns (a `MATERIALIZED`/`ALIAS` column cannot be named in an `INSERT`); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs. In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) +- **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`, and starts a tenant over on a fresh registry when a reload moves it to another address or database: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version and default timezone, and fires an `OnRefresh` hook after every successful refresh and before the registry reports itself loaded, so a loaded tenant is a bound one — `typelayer.Engine.Bind` is its only consumer (Key Design Decision #21) +- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else; `row` is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced for that stored record (via `typelayer.Table.Ingest`), `columns` names its positions (the table's insertable columns, or the narrower list a column-restricted role produces); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs with `typelayer.InsertSettings()` plus `async_insert=0` (it never loads the artifact, so an ingest-only process needs none). In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) - **`keyenc/`** — the one escaping composite keys are built from: `Escape` keeps `[A-Za-z0-9_-]` (exactly the tenant-id grammar, so a tenant id is its own escaped form) and writes every other byte as `%XX`, `Unescape` is `url.PathUnescape` (lenient: either hex case, and a byte left unescaped reads as itself, so a `%2D` an earlier build wrote still reads), `Join`/`AppendJoin` escape each field and put a separator between them (they panic on no fields, and on a separator the escaping could write or one outside ASCII) and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to and no field reaches a key unescaped: NATS subjects (`Join`/`Split` after the verbatim tenant) and the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — and the dedupe keys (`//`) use it; changing what it keeps orphans every stored key (an orphaned dedupe key lets a seen id through again), and on the shared backend, whose keys every process builds for itself, splits them between builds for the length of a rolling upgrade (a bump one build makes misses the entries the other filed, served until their TTL) - **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20), and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal, `ErrUnavailable` a broker that cannot be reached — both a `503`, with `Retry-After` `30` and `5`; `WithIdempotencyKey` makes a republish inside the queue's duplicate window a no-op), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). A broker whose ingest queue is split into units one consumer at a time owns implements `Sharded` too (`IngestUnits`, `ResetOrphaned`, `Unowned`; `ConsumerConfig.Units` narrows a consumer to some of them, and its consumer is a `Releaser` and a `Halter`, capped per unit by `ConsumerConfig.MaxHeld`, with `ErrConsumerMismatch` when an operator durable no longer fits), and `Message.OnSettled` runs a hook once, at the first ack or nak attempt, confirmed or not. Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, stream names, sequences, and ack floors are private to the implementations, whose subject tokens are escaped by the shared `internal/keyenc`: `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`, `deadletter.go`), which `internal/app` constructs and hands everything else as a `mq.Broker`, and `ExternalNATS` (`external.go`, `subject_nats.go`, `nats_topology.go`: an operator-owned cluster whose streams, durables and lease bucket it never creates, changes, purges or deletes; `lease.go` holds `coord.backend: nats`'s leases in that bucket), which `internal/app` constructs from the `mq.nats` block when `mq.backend` is `nats` ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every implementation passes the conformance suite in `internal/mq/mqtest` (`mqtest.Run`), which states the `Broker` contract as behavior; a new backend runs it from its own test, with `mqtest.Caps` only where its semantics legitimately differ - **`observability/`** — OpenTelemetry pipeline: `InitProvider` wires trace/metric/log providers via OTLP gRPC (each signal independently gated). A top-level `Prometheus` config block drives an optional `/metrics` scrape endpoint that runs independently of OTLP push — standalone (Alloy/Mimir scrape, no collector), alongside OTLP, or off. `NewLogger` produces a slog handler that fans out to stdout AND OTLP (stdout always 100%, OTLP sample-rate-aware). `TraceHandler` injects trace_id/span_id from active spans. `tracer.go` provides W3C trace context propagation over message headers (`InjectHeaders`/`ExtractHeaders` on a plain header map; `internal/mq` injects on every publish and extracts on the `Subscribe` path, so this package never sees a NATS type). @@ -46,7 +46,8 @@ Twenty internal packages under `internal/` (plus `internal/testutil/` for shared - **`policy/`** — Hasura-style access control, **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant splits by operation into `SelectPermissions` (columns, row `filter`, aggregations, the `max_*` limits) and `InsertPermissions` (columns, `check`) — so a field only one side honors does not exist on the other. `Evaluate()` resolves ONE operation and leaves the other side **nil** (`Select *ResolvedSelect` / `Insert *ResolvedInsert`), which every accessor fails closed on — nil is "not resolved", distinct from an empty side, which is "unrestricted" (what the admin return builds). Claim templating (`{{ jwt.claim.path }}`) resolves during that call. Policies come from `Source`, a `func() *Policy` read per call (`settings.Store.Policy` in production, `Static(p)` in tests) - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes -- **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) +- **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row via `typelayer`, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) +- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes` (the sole exception: `cmd/wavehouse/main.go` references `typelayer` itself). One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns, plus a `DEFAULT ''` per `_eq` check column — which is how column policy and auto-inject are answered with no Go-side record inspection. `Table.Ingest(format, body)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) - **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -55,7 +56,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 1. **Interface-first** — core behaviors are Go interfaces (`Cache`, `Deduplicator`, `Publisher`, `Subscriber`); standalone vs. future-clustered swap implementations. 2. **Bring Your Own Schema** — users create ClickHouse tables; WaveHouse discovers them via `system.columns` and never auto-migrates. -3. **Schema-driven ingest** — `POST /v1/ingest?table={table}` takes flat JSON, validated against the discovered schema (unknown fields rejected, types/nullability enforced). No envelope. The **declared `Content-Type` chooses the format and the bytes never do** (arity within the JSON family is still the body's): no declaration, one whose **media type** is unsupported or unparseable, a comma-bearing value that, as a whole, does not parse as one media type, or repeated lines that **disagree**, is a `415` decided *before* the body is read. A malformed *parameter* on a comma-free line never costs the request (`; charset=a; charset=b` still reads as its media type), and repeated lines are accepted only when they all resolve to the same **supported** format — two agreeing `text/csv` lines are still a `415`. A body declared NDJSON stays NDJSON whatever its bytes, so a bad line is a per-record error rather than a silent re-framing; the reverse (NDJSON sent as `application/json`) is deliberately **not** caught — record one, `200`, the rest ignored ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). Fail-closed — preserve it when touching `internal/api`. +3. **Schema-driven ingest: the body goes to ClickHouse's parser as-is** — `POST /v1/ingest?table={table}` never decodes a record. The **declared `Content-Type` chooses the format and the bytes never do** (`internal/api/content_type.go` is the whole table: the `application/json` and four NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, `; header=present` → `CSVWithNames` / `TSVWithNames`, `; header=absent` → the same with header detection off, a bare type → ClickHouse's default auto-detection); anything else — no declaration, an unsupported or unparseable media type, a comma-bearing value that does not parse as one media type, or repeated lines that **disagree** — is a `415` decided *before* the body is read, while a malformed *parameter* on a comma-free line never costs the request. The body decides exactly one thing, in `internal/api/ingest_framing.go`: the first non-whitespace byte picks array-vs-single inside `application/json`, which sets the response shape. A top-level array is re-framed in place (outer brackets and depth-1 commas blanked) so a compact array keeps per-record salvage; that rewrite must never run on a bare object or NDJSON, which it destroys. NDJSON sent as `application/json` is still not caught — record one, `200`, the rest ignored ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). CSV/TSV are positional over the table's wire columns. Fail-closed — preserve it when touching `internal/api`. 4. **Async ingestion** — ingest returns 200 after optional dedup + MQ publish; ClickHouse writes happen later via `StartIngestWorker`. NATS full → 503 + Retry-After. 5. **Per-tenant-table batching** — the worker groups events by tenant table (the tenant read off each message's `mq.Topic`), so one INSERT never mixes tenants and a batch invalidates its own tenant's cache namespaces; then it splits each batch by column list (`groupByColumns`), emitting one `INSERT INTO … (cols) FORMAT JSONCompactEachRow` per distinct list so a schema change mid-stream can't corrupt a statement. Each tenant table's batch is independent. 6. **Dead Letter Queue** — batch inserts ClickHouse **rejects** (isolated row by row; `chconn.Classify` == `Rejected` — a multi-row batch refused for its size, `chconn.Splittable`, is split row by row too) publish to the tenant's own dead-letter queue (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). A ClickHouse that cannot take the insert — unavailable, denied, or no verdict — never dead-letters a row, not even mid-isolation: the rows go back to the MQ with a delayed nak under a per-pool backoff — per table for a failure of one table (`chconn.TableScoped`: read-only, too many parts or mutations, a missing grant; `internal/ingest/backoff.go`), counted by `wavehouse_ingest_retries_total`. No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format`, or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. @@ -64,19 +65,20 @@ The invariant index — what must stay true. Full narrative and rationale live i 9. **Singleflight** — the cached read handlers coalesce concurrent misses (`x/sync/singleflight`) under the tenant-led cache key to prevent cache stampede, per tenant. 10. **Active Sweeper** — purges NATS messages that are both ACKed (written to CH) and older than the gap window; SSE gap-fill uses `DeliverByStartTime`, no in-process ring buffer. It runs only in the process holding the `sweeper` lease (`coord.RunElected`); that lease is not fenced: an overlap cannot lose ClickHouse data, since every sweep stops at the consumer's ack floor; it can only trim SSE replay history, and only when the two holders' settings views differ (one still reading a shorter `stream.gap_window_minutes`, or missing a tenant, after a reload the other has applied) — which fencing would not prevent either. Anything that does need exclusivity must check the term's `Token`. 11. **Hasura-style access control: fail-closed (security)** — `policy.IsAdmin` (role == `admin_role`, **exact case-sensitive**, default `"admin"`) is the single admin check, shared by `Evaluate`/`ResolveRole`/`Validate`/the `/v1/ops` gate/`RoleAllowed`. Empty/absent role matches nothing (no `"*"` wildcard); `Validate` rejects empty role keys; a `nil` policy (deleted) denies **everyone incl. admin** via a role — a total lockout for token-based callers, so recovery is writing `policies.json` and reloading, never an implicit admin grant (**exception:** the operator key's `auth.IsOperator` bit passes the `/v1/ops` gate even under a `nil` policy — a deliberate break-glass that can `POST /v1/ops/settings/reload` over HTTP, see #7). Over a nested settings directory the `/v1/ops` gate reads no policy at all — those routes reach every tenant, so the operator key alone passes and an admin-role token gets `403`; `api.NewRouter` decides that from the registry's shape, not from what was wired. `default_role` is the one sanctioned roleless exception (`ResolveRole` maps empty → it pre-eval); `default_role == admin_role` is permitted but dev-only and loudly warned (`policy.DefaultRoleGrantsAdmin`). Preserve when touching `internal/policy` (policy twin of #13; see #159). Detail: architecture.md § `policy/`. -12. **Structured queries: column authz fail-closed (security)** — `POST /v1/query?table={table}`: typed AST validated against schema, permission-enforced, timestamp-bucketed for cache, `DefaultMaxRows` (10,000) cap. Every column reference — projection, aggregation args, `filters`, `group_by`, `order_by`, `time_range` — is authorized inside `query.Build` (the single chokepoint that enumerates them all), so no clause can skip the role's `allow_columns`/`deny_columns` check (#223). A `select_all` read by a *column-restricted* role expands to its allowed columns via `policy.AllowedProjection`, never a bare `SELECT *`; *unrestricted*/admin roles keep `SELECT *` (`policy.RestrictsColumns` decides). Omitting `columns` selects nothing (`ErrEmptyProjection` → `200 []`); `["*"]` is the literal column `*` (schema-gated, not a wildcard); a table-granted role with no readable columns fails closed (`ErrNoReadableColumns` → `403`). Structured and live-stream (`stream.projectIndices`) reads share the one per-column decision `policy.IsColumnAllowed`, so column visibility can't drift. Row visibility has the same one-source guarantee (#319): `Evaluate` resolves a role's row-`filter` once (`resolvePredicates`), and both surfaces consume that single resolution — the query path renders it to SQL (`predicatesToSQL`), the stream evaluates it in memory per subscriber (`ResolvedPermissions.RowVisible`, whose type-aware comparison fails closed on anything it can't prove about the ingested payload — `policy.ColumnSpec`, with `DateTime`/`DateTime64` operands compared as instants through the ingest grammar (`discovery.Column.TimeParser`) and claim constants rendered canonically and digit-exact by the one shared rule `policy.CanonicalScalar` (#457 — which also refuses a float64 at/past 2^53 rather than match a neighboring ID, and whose ok=false — an absent claim, a structured value, no canonical form — makes the predicate match no rows on BOTH surfaces: `1 = 0` in SQL, every row withheld in memory); numeric comparison runs in the column's STORAGE domain (`policy.NumericSpec`, classified by `discovery.NumericStorageOf` — Float width rounding, Decimal scale truncation, integer exactness, both operands narrowed as ClickHouse narrows stored value and bound constant, out-of-range operands refused rather than modeled; the `tests/integration` differential oracle holds in-range verdicts equal to a live ClickHouse's and the never-admit-where-SQL-hides direction for the refused out-of-range ones); an event whose insert later fails into the DLQ is the one residual payload-vs-stored asymmetry, documented in the access-control enforcement caution) — so row visibility can't drift either. Preserve when touching `internal/query` or the structured-query handler. Detail: architecture.md § `query/`. +12. **Structured queries: column authz fail-closed (security)** — `POST /v1/query?table={table}`: typed AST validated against schema, permission-enforced, timestamp-bucketed for cache, `DefaultMaxRows` (10,000) cap. Every column reference — projection, aggregation args, `filters`, `group_by`, `order_by`, `time_range` — is authorized inside `query.Build` (the single chokepoint that enumerates them all), so no clause can skip the role's `allow_columns`/`deny_columns` check (#223). A `select_all` read by a *column-restricted* role expands to its allowed columns via `policy.AllowedProjection`, never a bare `SELECT *`; *unrestricted*/admin roles keep `SELECT *` (`policy.RestrictsColumns` decides). Omitting `columns` selects nothing (`ErrEmptyProjection` → `200 []`); `["*"]` is the literal column `*` (schema-gated, not a wildcard); a table-granted role with no readable columns fails closed (`ErrNoReadableColumns` → `403`). Structured and live-stream (`stream.projectIndices`) reads share the one per-column decision `policy.IsColumnAllowed`, so column visibility can't drift. Row visibility has the same one-source guarantee (#319): `Evaluate` resolves a role's row-`filter` once (`resolvePredicates`, exposed via `Predicates()`), and both surfaces consume that resolution — the query path renders it to SQL, the stream compiles it through `internal/typelayer` and evaluates it per subscriber. Only a definite true admits; error, decline, drift and an unavailable engine all withhold, counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. A claim compared against an integer column goes through one strict round-trip cast in both renderers, so a claim that does not fit the column matches nothing instead of wrapping. The one residual payload-vs-stored asymmetry is an event whose insert later fails into the DLQ, documented in the access-control enforcement caution. Preserve when touching `internal/query` or the structured-query handler. Detail: architecture.md § `query/`. 13. **Named query pipes: fail-closed (security)** — pre-defined SQL templates (Tinybird-style) with param binding + caching — reads only: a pipe whose SQL `IsMutation` classifies as a write bypasses the cache and singleflight, since a cached or coalesced write is a dropped one (#386); `GET/POST /v1/pipes/{name}` sit outside `RequireAdmin`, so per-pipe `allowed_roles` is the *only* execute-path gate, via `policy.RoleAllowed`: exact allowlist membership (no `"*"`), admin always passes, empty/absent role and empty-string entries authorize nobody, and no `allowed_roles` → admin-only. Preserve and exercise via `testutil.RunRoleMatrix` / `StandardRoleMatrix` (see #159). Detail: architecture.md § `pipes/`. 14. **TypeScript SDK** — `@wavehouse/sdk`: typed query builder, real-time SSE over `fetch`, live queries (incrementable/decomposable/poll aggregation), codegen CLI. Exactly one runtime dependency — `eventsource-parser` (SSE framing, itself dependency-free); adding a second needs the same scrutiny the first got. The canonical client (see §SDK Sync). 15. **Observability invariants** — stdout always 100% (sampling is OTLP-push-only); WARN+ERROR always export at 100% (a non-configurable floor — don't expose it); gRPC OTel exporters dial lazily so an unreachable collector never blocks startup; the OTel Prometheus exporter uses a **private** `prometheus.Registry`. The OTLP endpoint/TLS/custom-CA/mTLS/headers are delegated to the OpenTelemetry SDK's standard `OTEL_EXPORTER_OTLP_*` env vars — `InitProvider` passes **no** endpoint/header options. Known gap, intentionally not patched in WaveHouse app code: the pinned gRPC logs exporter (`otlploggrpc` v0.19/v0.20) ignores the env TLS-cert vars, so a custom/private CA and mutual TLS apply to traces/metrics but **not** the logs signal (public-CA/system-roots TLS and plaintext still work for logs) — upstream bug open-telemetry/opentelemetry-go#6661. A malformed `OTEL_EXPORTER_OTLP_HEADERS` is logged and skipped by the SDK (fail-soft), not fatal. Preserve when touching the logger/sampler/provider. Detail: architecture.md § `observability/`. 16. **Bearer-token-only CORS posture (security)** — Bearer JWT on every request, no cookies/sessions; `corsMiddleware` deliberately **never** emits `Access-Control-Allow-Credentials` (not needed, and `*` + credentials is a spec violation browsers reject). `cors.allowed_origins` (settings directory, per tenant: a tenant route is decorated from the list of the tenant it names, everything else from tenant `0`'s — `corsOrigins`) controls who can *read* responses, not cookie scope; CSRF protection is structural. Don't reintroduce cookie auth or `Allow-Credentials` without a design discussion — answers GitHub #29/#30. Code: `internal/api/router.go`. 17. **Non-fatal boot** — schema-discovery failure on boot is non-fatal: `internal/app` records an `api.BootState`, binds `:8080`, serves 503 on `/livez`/`/readyz` with the diagnostic, and retries via `SchemaRegistry.RetryRefresh` (jittered backoff 2s → 60s), per tenant over a nested directory: `/livez` is 503 while no tenant has completed a first discovery, then sticky 200, and a tenant's outage after that is its log line and counter, never a probe failure. Until a tenant's first discovery its table lookups are a 503 with `Retry-After`, not a 404. Bounds supervisor restart loops. 18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention; `/readyz` pings every open ClickHouse pool at once and is ready at the first answer, 503 naming each when none answers); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. -19. **Canonical timestamp wire form (fail-open at ingest)** — the HTTP ingest handler rewrites every top-level `DateTime`/`DateTime64` column value it can parse to RFC 3339 UTC (`discovery.CanonicalizeTimestamps`; per-column precision + zone precomputed at schema refresh) after validation + policy checks and **before** the NATS publish, so the one payload every consumer shares — SSE subscribers, the ClickHouse insert, the DLQ — carries the same spelling `/v1/query` renders: live and query reads can't drift on the instant (#372). Zone-less inputs are read in the column's declared zone, else the discovered server default — ClickHouse's own rule, so the spelling changes but never the instant. Deliberately **fail-open**: an unparseable value or unresolvable zone (no tzdata embedded — never a failed refresh, never a silent UTC reinterpretation, which would move instants) publishes verbatim; ingest must not reject a record over its timestamp spelling — fail-closed enforcement belongs to the stream row-filter (#381). Don't re-spell timestamps downstream. Preserve when touching `internal/discovery`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `discovery/` + §Ingest Path; the exact spelling spec (truncation, zero-trimming, `Z`-only) lives in api.md §Timestamp canonicalization — keep it in sync with `canonicalTimestamp`. +19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer.Table.Ingest`, via chtypes), in the column's declared zone else the server's default (`"2026-06-21 04:00:00.123"`, never RFC 3339's `Z` suffix), and `/v1/query` and pipes are rendered by the same ClickHouse in the same spelling — there is no separate WaveHouse rewrite step to keep in sync, so live and query reads can't drift on spelling *or* instant (#372) while the process's chtypes zone equals the zone the server renders in. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. 20. **Sealed MQ boundary** — only `internal/mq` imports NATS/JetStream (`github.com/nats-io/…`), enforced by the `depguard` rule in `.golangci.yml`, so `make lint` fails on a leak in every package it builds (the `integration`-tagged files under `tests/` are outside lint's build context — keep them clean by convention, through `mq.Broker`). A test outside `internal/mq` that needs a real NATS server goes through `internal/mq/natstest`, which stands one up from the shipped `deployments/nats` files and hands back a URL and passwords, never a NATS type. The boundary is semantic as well: everything else addresses events by `mq.Topic` and states intent through mq-owned interfaces (`Publisher`, `Consumer`, `DeadLetterer`, `Purger`, `Replayer`, …), and never builds a subject, names a stream, or reasons in sequences — so a subject, stream, or broker change lands in one package ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 4; story 5's tenant token landed there alone — `Topic.Tenant`, first in every subject). Don't add a raw accessor (`JetStream()`, `NatsConn()`, `GetServer()`) back, and don't hand-build `"ingest."`/`"dlq."` subjects outside `internal/mq` — widen the mq surface with an intent-level method instead. +21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503`, the stream withholds every row with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}`. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. ## Code Conventions -- **Go 1.26**, strict formatting (`gofumpt`, enforced by CI) +- **Go 1.27**, strict formatting (`gofumpt`, enforced by CI); cgo enabled (`internal/typelayer`'s chtypes dlopen shim needs a C toolchain + glibc) - **Structured logging** with `log/slog` (JSON handler), through the default logger: call `slog.InfoContext(ctx, …)` and its siblings (the context carries the trace ids the handler stamps) rather than taking a `*slog.Logger` parameter or field. `cmd/wavehouse` and `internal/app` install the default; tests silence or capture it with `internal/testutil/logtest` (a capturing test must not call `t.Parallel()`) - **Chi v5** for HTTP routing - **Error handling**: Return errors, don't panic. Wrap with `fmt.Errorf("context: %w", err)`. @@ -437,7 +439,7 @@ internal/chsql/ → Shared ClickHouse SQL helpers (identifier quoting + internal/config/ → Configuration structs + loader internal/coord/ → Leases with fencing tokens (interface, in-process Local, RunElected, coordtest conformance suite) internal/dedupe/ → Optional deduplication (Reserve/Commit/Release interface; Pebble, DynamoDB) -internal/discovery/ → ClickHouse schema introspection + ingest validation +internal/discovery/ → ClickHouse schema introspection (system.columns/system.tables, server version + timezone) internal/ingest/ → Batch buffer with DLQ + Active Sweeper (NATS message lifecycle) + shard claims over a shared queue internal/keyenc/ → One escaping for composite keys (NATS subject tokens, cache keys, dedupe keys) internal/mq/ → MQ boundary (the only NATS/JetStream importer: owned message/consumer/stream types, the embedded server and the external-NATS broker; mqtest/ is the Broker conformance suite; natstest/ stands up NATS as an operator deploys it, for tests outside the package) @@ -447,6 +449,7 @@ internal/policy/ → Access control policies (types, evaluation, Source) internal/query/ → Structured query AST + SQL builder internal/settings/ → Settings directory (validate, adopted snapshot + reload, watcher, embedded seed) internal/stream/ → SSE fan-out (event Hub: project once per role, Subscriber outbound queue, Bucket fan-out, keepalive Heartbeater wheel) +internal/typelayer/ → In-process ClickHouse parser (chtypes): ingest validation/coercion + row-level-security compilation internal/tenant/ → Tenant id (type, grammar, reserved default, request header name) internal/testutil/ → Shared test helpers (mocks, JWT + schema helpers; logtest/ captures or silences the default logger; cachetest/ is the conformance suite every cache.Cache backend runs; mutationtest/ holds the shared write-classifier cases; storedir/ is the embedded broker's store directory in tests, removed once late consumer-state writes land) tests/ → Integration & E2E tests diff --git a/CHANGELOG.md b/CHANGELOG.md index 10229066..8eb238cb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -35,6 +35,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **A nested settings directory serves one tenant per folder, failing closed per tenant** (`internal/settings/{tree,registry,store,validate,watch}.go` (`tree.go` new, + tests), `internal/api/{tenant,router,pipes,settings}.go`, `internal/app/{app,wire}.go`, `cmd/wavehouse/validate.go`, `clients/ts/src/{pipes,settings,types,index}.ts`, `tests/e2e/sdk/admin.test.ts`, `docs/src/content/docs/{deployment,api,architecture}.md`, `docs/src/content/docs/{settings-directory,reverse-proxy,access-control}.mdx`, `docs/src/content/docs/sdk/{admin,pipes,reference,streaming}.md`): story 2 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files — the same findings byte for byte, the same boot refusal, the same keep-previous on a rejected reload, the same watcher and `SIGHUP`, the same `wavehouse validate` exit codes. `settings.Validate` now reads the directory's shape off its entries — any of the four file names makes it flat, otherwise a folder makes it nested — and checks either one: each tenant folder goes through the per-directory checks (now `ValidateDir`), its name through the tenant-id grammar, and its findings carry the folder (`acme/policies.json`). The shapes never mix: a flat directory rejects a folder and a nested one rejects a loose file. `settings.Open` returns the `Registry`, which now owns `Reload`, the new `ReloadTenant`, the `AfterAdopt` hooks (handed the tenants a reload adopted), and the watcher; a `Store` is a passive holder of one tenant's document. A nested directory fails closed per tenant, at boot and on reload alike: a folder with an error finding stops its tenant being served — the tenant routes answer a bare `503 {"error": "tenant settings are invalid"}`, since they resolve before authentication and the findings quote the settings — while every other tenant carries on, with no previous-snapshot fallback; a request already admitted finishes on the document it started with, and the tenant keeps one `*Store` across the rejection. A whole-directory reload mirrors the folders (a new one is served, a removed one is a `404`), and a finding about the directory itself — a loose file, an unreadable directory, a changed shape — refuses boot and rejects a reload whole, leaving every tenant as it was. A nested directory gets no watcher; `SIGHUP` reloads the whole tree in both shapes. `POST /v1/ops/settings/reload` and the admin pipe reads (`GET /v1/ops/pipes[/{name}]`) take an optional `?tenant=`, parsed strictly — a query string that does not parse, or an empty, repeated, or malformed `tenant`, is a `400`, never a read of the default tenant — with `404` for an unknown tenant; absent means the whole directory on the reload and tenant `0` on the reads. The reload response keeps its shape: over a nested directory `adopted: false` with a `422` can mean adopted in part, the rejected folders being the ones with an error among their `findings`. Over a nested directory the `/v1/ops/*` gate admits the operator key alone — those routes reach every tenant — and an admin-role token gets `403`; `api.NewRouter` decides that from the registry's shape, whatever policy source was wired. Booting a nested directory with no `auth.operator_key` therefore leaves `SIGHUP` as the only reload, and boot warns about it; a whole-directory reload that drops a tenant names it in the log; and `Registry.Watch` refuses a nested directory itself. The resources a process still has one of (ClickHouse connection, dedupe store, MQ byte budget, auth verifier) follow the settings tenant `0` last adopted, their reload hooks running only when tenant `0` is adopted, so another tenant's reload never moves them and a `0` folder that a reload rejects or removes leaves all of them as they were; a nested directory without a `0` folder boots with them unconfigured (no ClickHouse address, `/livez` degraded) and says so once at boot, and the async paths (ingest worker, sweeper, stream hub, schema refresh) stay wired to tenant `0` until story 5. The SSE keepalive wheel is the one shared resource that weighs every tenant: it runs at the shortest `stream.keepalive_interval` among the tenants being served ([#597](https://github.com/Wave-RF/WaveHouse/issues/597) tracks honoring each tenant's own). The SDK's `wh.pipes.list()`, `wh.pipes.get()`, and `wh.settings.reload()` take a `tenant` option (the new `OpsRequestOptions`) sent as `?tenant=`, since `options.headers` cannot set a query parameter. - **Requests resolve to a tenant before authentication, and the tenant is threaded through every settings read** (`internal/tenant/` (new, + tests), `internal/settings/registry.go` (new, + tests), `internal/api/tenant.go` (new, + tests), `internal/api/{router,ingest,structured_query,pipes}.go`, `internal/ingest/{worker,sweeper}.go`, `internal/stream/hub.go`, `internal/discovery/discovery.go`, `internal/app/{app,wire}.go`): story 1 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a deployment that sends no tenant header. `internal/tenant` defines the id — a validated string (letters, digits, `_`, `-`; at most 64 bytes, so it is safe as a folder name and as an MQ subject token), the reserved default `0`, and the `X-Tenant-ID` header name — and imports nothing from the rest of the repository. `api.TenantMW` runs ahead of the auth middleware on every `/v1` route outside `/v1/ops/*`: an absent or empty header is tenant `0`, a malformed id or a repeated header is a `400`, a well-formed id the new `settings.Registry` does not hold is a `404`, and the resolved `*settings.Store` rides the request context. Every answer from the middleware, the `400` and `404` included, carries `Vary: X-Tenant-ID` so a shared cache can't replay one tenant's response to another. The registry holds the one store `settings.Open` adopted, keyed `0`. Handlers read the store once and pass it down as an argument — the ingest, structured-query, and pipe getters (`PolicySource`, `DedupeSettings`, `bucketSecs`, `defaultMaxRows`, the pipes source) now take it as a parameter — and a tenant route reached without a resolved tenant answers `500` rather than fall back to one. The ingest worker, sweeper, stream hub, and schema registry are constructed with a `tenant.ID` (`tenant.Default` in `internal/app`) and their settings getters take it; a tenant the registry does not hold is logged and read as the getter's zero value, with two fail-safes — the worker's DLQ switch reads as on (an unreadable message is parked, never dropped) and the schema auto-refresh keeps its cadence rather than hand `time.NewTicker` a zero interval. The probes, `/version`, the metrics path, and `/v1/ops/*` stay tenant-exempt, with the ops tree behind the auth middleware and the admin gate exactly as before; the admin pipe reads (`GET /v1/ops/pipes[/{name}]`) serve the default tenant. `X-Tenant-ID` joins the CORS `Access-Control-Allow-Headers` list so a browser client can send it; the SDK needs no change (`options.headers`). +- **Ingest accepts CSV and TSV bodies** (`internal/api/content_type.go`, `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `docs/src/content/docs/api.md`): `Content-Type: text/csv` and `text/tab-separated-values` are read by ClickHouse's own `CSV`/`TSV` readers. `text/csv; header=present` and `text/tab-separated-values; header=present` read `CSVWithNames` / `TSVWithNames` (the first line names the columns, in any order; an omitted column takes its `DEFAULT`; an unknown or duplicate name is code 117 for the whole request). `header` is RFC 4180 §3's optional parameter, mapped three ways onto ClickHouse: `header=present` is `CSVWithNames`, `header=absent` is strictly positional (`input_format_csv_detect_header=0` / `input_format_tsv_detect_header=0`), and a bare type is ClickHouse's default reading with header auto-detection on, so a first line that spells the column names is consumed as a header. Positionally, the fields are the table's wire columns — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and every one of them must be present, in that order. An empty CSV field (and `\N` in TSV) takes the column's `DEFAULT`; too few fields is code 27, too many is 117, and under `header=absent` a header line is one record that fails to parse with code 27. Always the batch response shape. - **Schema discovery captures each table's DDL, its columns' ordinals and default expressions, and the server version** (`internal/discovery/discovery.go`, `internal/testutil/testutil.go`): `Column` gains `DefaultExpression` and `Position` (both from a widened `system.columns` select), `TableSchema` gains `DDL` from `system.tables.create_table_query`, and `SchemaRegistry` gains `ServerVersion()` from a `SELECT version()` probe next to the existing `SELECT timezone()`. Groundwork for the native type layer, captured on the same refresh as the columns so a stale version cannot outlive the schemas it describes. That is a publication guarantee, not a same-server one: `chconn.Manager` resolves the connection per call, so a reload changing `clickhouse.addr` mid-refresh can still pair a version from one server with schemas from another — narrow, and self-correcting on the next refresh. `DDL` is `json:"-"` and does **not** appear in `/v1/ops/schema`: that endpoint marshals `TableSchema` straight to the client, and an external-engine table (S3, MySQL, PostgreSQL, Kafka) renders its wiring there unconditionally — endpoint, bucket or host, database, username, S3 access key id. ClickHouse masks the password itself as `[HIDDEN]` from ~23.9 (verified on 26.7.3), so the exposure is the topology rather than the secret — except on an older server, or one with `display_secrets_in_show_and_select` enabled. `position` and `default_expression` are additive fields in the response. A table listed in `system.tables` with no `system.columns` rows is skipped rather than published column-less, and both new queries fail the refresh on error exactly as `timezone()` and `system.columns` do — callers keep the prior cache and retry. @@ -58,6 +59,15 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Logging goes through the `slog` default logger; no constructor takes a `*slog.Logger` anymore** (`internal/mq/embedded.go`, `internal/api/{ingest,pipes,structured_query,dlq,settings,errors,router}.go`, `internal/auth/auth.go`, `internal/discovery/{discovery,timestamp}.go`, `internal/ingest/{sweeper,worker}.go`, `internal/settings/{store,watch}.go`, `internal/chconn/chconn.go`, `internal/config/persistence.go`, `internal/app/wire.go`, `internal/testutil/logtest/` (new, + tests), `internal/testutil/testutil.go`): the general-notes refactor of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) and the cleanup deferred from [#586](https://github.com/Wave-RF/WaveHouse/pull/586), which left `internal/mq` logging half through an injected logger and half through the default. The logger parameter or field is gone from `mq.NewEmbedded`, `api.NewIngestHandler` / `NewPipesHandler` / `NewStructuredQueryHandler` / `NewDLQHandler` / `NewSettingsHandler`, `api.RequireAdmin` and `api.Dependencies.Logger`, `auth.NewAuthenticator`, `discovery.NewSchemaRegistry`, `ingest.NewSweeper` and the ingest worker, `settings.Open`, `chconn.Open` (whose field was never read), and `config.WarnIfFreshDataDir` / `LogStorageInitError`. Call sites use the context-aware calls (`slog.ErrorContext(ctx, …)`) wherever a context is in scope, so the trace handler can stamp them. Two visible differences: the ingest worker's lines no longer carry `component=ingest_worker`, and the auth middleware's operator-key audit lines and the settings reload lines are no longer skippable by passing a nil logger (only tests did). Tests reach log output through the new `internal/testutil/logtest`: `Silence()` from a package's `TestMain`, and `Capture(t, level)` for a test that asserts on log lines — which therefore runs serially, since the default logger is process-wide. `testutil.NopLogger` is removed. - **The process wiring moves out of `main.go` into `internal/app`** (`internal/app/` (new: `app.go`, `wire.go`, + tests), `cmd/wavehouse/main.go` (+ tests), `tests/integration/setup_test.go`, `.testcoverage.yml`, `.github/labeler.yml`): `app.New` builds every component from the boot config and the settings directory, `Run` drives the long-lived ones — ingest worker, sweeper, hub bridge, keepalive wheel, schema refresh, SIGHUP and the directory watcher, the API server and the Prometheus sidecar — under one `errgroup` until the signal context is cancelled or one of them fails, and `Close` releases what `New` opened in reverse order. Each component is wired in one place — what it opens, what it loops, what it releases — with the settings store handed to its wiring function whole, so the per-tenant registry ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)) lands there rather than in `main`. `main.go` shrinks to argv dispatch, the logger, `config.Load`, `CheckDataDir`, `app.New`, `app.Run`; `run(ctx)` takes the context `main` cancels on the first `SIGINT`/`SIGTERM`, so it is unit-tested end to end and the per-suite coverage exclude for it is gone. The integration suite boots through the same `app.New` against its testcontainer (the seed settings patched to the container, `default_role` set to the admin role) instead of a hand-built handler subset that had drifted from the binary. Behavior is unchanged except for the stop, which is now bounded end to end in three phases whose budgets add rather than multiply — `server.shutdown_timeout` for the drain, then fixed 5s and 3s for the release and the telemetry flush, so the worst case is the timeout plus 8s: `Run` drains the ingest worker and the API server's in-flight requests — the process could previously exit while the worker drain was still in flight — while open SSE streams are ended the moment the drain begins (a stream is a connection to close, not work to wait for; the client reconnects via `Last-Event-ID`) instead of holding the stop for the whole timeout; then `Close(ctx)` releases the stores under a context of its own, so a remote store's close can give up at the deadline rather than hang the exit, and finally flushes telemetry on a separate short budget so the flush that reports on the stop is never starved by a slow close. A settings reload caught mid-hook by the stop gives up with it. A second `SIGTERM`/`SIGINT` during the stop abandons it and exits non-zero, and `SIGHUP` is ignored once a stop has begun (it was briefly fatal: the reload loop's `signal.Stop` restored the default disposition at the start of the drain). The `mq.max_bytes_gb` reload hook bounds its JetStream calls to ten seconds, and when the DLQ resize fails it rolls the ingest stream back under a budget of its own instead of the one that just expired. `deployments/compose/standalone.yaml` sets `stop_grace_period` to cover all three phases, and the deployment docs gain a [Stopping](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md#stopping) section. Closes [#140](https://github.com/Wave-RF/WaveHouse/issues/140); story 0 of #583. +- **The chtypes SDK is `go/v0.4.0` (ABI revision 6)** (`go.mod`, `chtypes.lock`, `scripts/fetch-chtypes.sh`, `.github/actions/setup-env/action.yml`): the lock pins the revision-6 26.6.8.7 build (`b1790767905`) on darwin-arm64, linux-amd64 and linux-arm64, and must be regenerated whenever the SDK's ABI revision changes. The default artifact cache moved to `~/.cache/chtypes/artifacts/abi6/-`, so the first fetch after upgrading downloads again (explicit `--dest` / `CHTYPES_REGISTRY` directories, as the Docker images use, are unaffected); CI's cache key and path carry the revision. Insert `check` clauses now cost one parse instead of two: they are judged inside the same `RowsExportWith` call that validates the body, through a compiled row filter, and the five ingest formats (JSON family, CSV, TSV, CSV and TSV with `header=present`) all take that path. + +- **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `builds:` and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes --notes-start-tag "$(git describe --match 'v*')"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. + +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.4.0, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/hub.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, beside the string `code` class the other error bodies carry; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` results — carry ClickHouse's own rendering** (`"2026-06-21 04:00:00.123"`, in the column's zone) instead of the RFC 3339 `Z`-suffixed form WaveHouse used to canonicalize to, by construction rather than by a rewriting step (closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5`, and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. + +- **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, without the columns it may not write, instead of walking a decoded record's keys. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a checked column behaves exactly like omitting it. + +- **WaveHouse now requires cgo, and supported platforms narrow to darwin/arm64, linux/amd64, linux/arm64** (BREAKING; `go.mod`, `scripts/build.sh`, `.goreleaser.yaml`, `deployments/Dockerfile`, `deployments/Dockerfile.goreleaser`, `Makefile`, `internal/config/config.go`, `config.yaml`, `cmd/wavehouse/main.go`, `chtypes.lock` (new), `scripts/fetch-chtypes.sh` (new), `.github/actions/setup-env/action.yml`, `.github/workflows/ci.yml`): the native type layer above needs cgo for `dlfcn` (no C library linked, no header). cgo is now unconditional: `CGO_ENABLED=0` is gone from every build path, and the `make audit-cgo` target that policed the old no-cgo build has been **removed** along with it (`make binary-analysis` is now `size` + `deadcode`). Because chtypes publishes artifacts only for darwin-arm64, linux-amd64 and linux-arm64, **Windows, FreeBSD and darwin/amd64 builds are discontinued** — `.goreleaser.yaml`'s matrix drops from 8 targets to 3, and the release archives/checksums/GHCR image narrow to match. The runtime image moves from an Alpine/musl builder + `distroless/static` to `golang:1.27-bookworm` (glibc, ships gcc) + `distroless/cc-debian12` (glibc + libstdc++, which the SDK's shared library needs) and bakes the pinned chtypes artifact into the image at `/opt/chtypes/artifacts` via a new `chtypes.lock` (exact file + sha256 per platform/line) and `scripts/fetch-chtypes.sh --frozen` wrapper, so the container has no first-request download. `go.mod` moves to `go 1.27`. Only processes running the `api` role load the artifact, so an ingest-worker-only or sweeper-only process needs none installed; the glibc requirement is 2.34 or later. New boot config: `clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY` lets an operator point at an explicit registry directory instead of the SDK's own search path (the shipped image instead sets the SDK's own `CHTYPES_REGISTRY` env var directly). CI's `unit`/`integration`/`e2e` jobs fetch and cache the pinned artifact (`setup-env`'s new `chtypes` input) and set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` so a missing artifact fails the job instead of silently skipping the chtypes-backed tests. `GOLANGCI_LINT_VERSION` bumped `v2.11.4` → `v2.13.2`: the `go 1.27` bump panics `v2.11.4`'s type checker on every package; `v2.13.0` is the oldest release whose changelog claims go1.27 support, but it panics in this tree for a different reason (`nilness`/`honnef.co/go/tools@v0.8.0-rc.1` crashing while analyzing a third-party dependency), fixed once that dependency moves past its release candidate in `v2.13.1`. The cross-toolchain approach this bullet originally described was replaced before landing — see the release-pipeline entry above. - **Boot refuses an unbound `WH_*` environment variable and an unusable `data_dir`** (BREAKING; `internal/config/check.go` (new, + tests), `internal/config/{config,persistence}.go`, `cmd/wavehouse/main.go`, `docs/src/integrations/diagram-png.mjs`): the environment half of the strict YAML loader. `config.Load` now errors, naming every offender, on a `WH_*` variable that no `Config` field binds — the two variables read outside the struct, `WH_CONFIG` and `WH_LOG_LEVEL`, are exempt — `WH_DEDUPE_ENABLED=true` left in a compose file from before the settings-directory move, or a misspelling, was set, ignored, and believed. **An existing deployment that still exports a variable this release moved to the settings directory stops booting until it is unset**; the upgrade runbook in `deployment.md` gains that audit. Only the `WH_` prefix is checked, since the environment always carries unrelated names; the one outside source that shares it — Kubernetes service-link variables for a Service named `wh` or `wh-*` — is named in the error with the `enableServiceLinks: false` remediation, and the docs build's opt-out knob is renamed from `WH_SKIP_DIAGRAM_PNG` to `DOCS_SKIP_DIAGRAM_PNG` so an exported one no longer refuses a local boot. Right after `Load`, before ClickHouse or the settings directory are touched, `config.CheckDataDir` probes `data_dir` and refuses boot on any of: an empty or blank value (reachable through `WH_DATA_DIR=`), refused outright since the ancestor walk would otherwise fall back to the working directory and NATS and Pebble state would land under it; a path that exists and is not a directory; a dangling symlink at `data_dir` or any component above it (the walk to the nearest existing ancestor uses `Lstat`, so a failed mount is not skipped over as "does not exist" and passed in an unrelated directory); and a directory the process cannot write to — or, when it does not exist, an unwritable nearest ancestor — probed by creating and removing one temp file. So an unusable `data_dir` refuses boot before schema discovery rather than after it; a permission denial — on the probe, or on reaching the path at all through a parent without search permission — carries the UID-65532 remediation (a bind mount owned by root is the typical cause), and that hint string is now shared with `LogStorageInitError`. `EnvConfig` and `EnvLogLevel` join `EnvSettingsDir` as the exported names for the process-level variables. Boot is the validator for the non-hot-reloadable half — there is no dry-run subcommand, by decision on #530: boot config only takes effect through a restart, so the restart is where it is checked, and the docs say so. Closes #530. @@ -83,10 +93,14 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The landing page's live demo now reads from the stats deployment's new WaveHouse Cloud backend** (`docs/src/components/LiveDemo.astro`, `docs/scripts/screenshot.mjs`): the GitHub-activity dogfood deployment behind the hero panel (Wave-RF/WaveHouse-Stats) moved off its self-managed AWS infrastructure onto WaveHouse Cloud, so `BASE_URL` — the origin `@wavehouse/sdk` queries in the visitor's browser — points at `https://iefrrvavd5akvphk7pq3.wavehouse.app` instead of `https://stats.wavehouse.dev`, ahead of the AWS stack being torn down. The `PUBLIC_WAVEHOUSE_STATS_URL` build-time override is unchanged, so a fork or staging docs build still redirects the panel without a code edit. **`DEMO_HOST` deliberately stays `stats.wavehouse.dev`** — the demo *site* is still served there and is still what the panel's chrome label and "Full demo" link should show; the migration splits the site from the API origin behind it, and the two constants now carry comments saying so. Verified against the new deployment before the switch: all five pipes the panel reads (`gh_summary`, `gh_activity_recent`, `gh_events_per_minute`, and the pre-#19 `gh_stars_total` / `gh_forks_total` fallbacks) return `200` with the same row shapes, the structured-query backfill fallback (`POST /v1/query?table=gh_events`) matches its old-backend response byte for byte, `GET /v1/stream?table=gh_events` opens an SSE stream, and CORS is unchanged (`Access-Control-Allow-Origin: *`, `X-Cache` exposed) so the cross-origin browser reads keep working from the docs site. The new backend is already the live ingest target — it reported more recent events than the old one at cutover (5,175 vs 5,038 over 7d) — which is the other half of why the panel had to follow it. `screenshot.mjs`'s `networkidle` note is retargeted to "the stats demo backend" rather than naming a host it no longer connects to. +- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is ClickHouse's spelling (`"2026-06-21 04:00:00.123"`, in the column's zone, no offset) where it was RFC 3339 UTC, and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list binds as one `Array(String)` parameter, with a size and count check that answers `400` before the request is sent. The RFC 3339 filter-value rewrite is deleted, so a `"…Z"` value goes to the server verbatim and works as far as the connected server's version reads that spelling. Failure classification is unchanged — the same `code`/`retryable` table of the query paths — and each tenant's reader connections are capped at its `max_open_conns`. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. + - **WH001 (no hard-wrapped prose) now applies to every tracked Markdown file, with no carve-out** (`.github/.markdownlint.json` (deleted), `.claude/.markdownlint.json` (deleted), `.claude/skills/integration-astro-view-transitions/` (deleted), `.markdownlint-cli2.jsonc`, `.github/workflows/README.md`, `.claude/skills/pm-triage/references/routine.md`, `AGENTS.md`, `scripts/docs-prose.sh`, `.github/prompts/docs-review.md`, `docs/src/content/docs/claude-code.md`, `docs/src/content/docs/development.md`, `.claude/agents/docs-reviewer.md`): two path-scoped configs had switched WH001 off under `.github/` and `.claude/` ever since [#489](https://github.com/Wave-RF/WaveHouse/pull/489) introduced the rule — baked in from the start rather than added in response to a discovered problem — which left the repo documenting the rule three ways and disagreeing with itself: `CONTRIBUTING.md` promises contributors `make lint` enforces it *everywhere*, while `AGENTS.md` and the `.markdownlint-cli2.jsonc` header wrote up the carve-out. Not theoretical: on [#520](https://github.com/Wave-RF/WaveHouse/pull/520) a reviewer correctly flagged a hard-wrapped bullet in `.github/workflows/README.md`, an agent pointed at `"WH001": false` for that path and pushed back, and the reviewer recorded a *learning* never to flag WH001 there — the wrong invariant, learned off the wrong side of the contradiction ([#521](https://github.com/Wave-RF/WaveHouse/issues/521)). Both configs are deleted — each held nothing but the override, so the root `.markdownlint.json` governs again — and the 51 hard-wrapped paragraphs they were hiding are joined: 41 in `.github/workflows/README.md` and 10 in `.claude/skills/pm-triage/references/routine.md`, mechanical joins with no wording changed and every fenced block, table row, and heading byte-identical either side of the reflow. Deleted with them: the wizard-installed PostHog skill at `.claude/skills/integration-astro-view-transitions/` — 9 files, ~1,456 lines, including an 809-line `EXAMPLE.md` copied wholesale from `PostHog/context-mill`. Its integration job finished in [#277](https://github.com/Wave-RF/WaveHouse/pull/277), nothing in the repo calls it, and the docs-site setup it once described is documented where it belongs — in `docs/src/components/PostHog.astro` and this file. Keeping unowned third-party prose in the tree means content that drifts silently on every upstream bump and that nobody here reviews; it was also the single file that would have needed a special-case lint exclusion, so removing it is what lets WH001 apply with **no exception at all** rather than one documented one. Its two inventory rows in `claude-code.md` go with it, as does the now-dead `docs/posthog-setup-report.md` entry in the `scripts/docs-prose.sh` denylist (the wizard's other artifact, deleted back in [#502](https://github.com/Wave-RF/WaveHouse/pull/502)) and the copies of that denylist in `AGENTS.md` and `.github/prompts/docs-review.md`, which the script's header requires be kept in lockstep. Review of the change then turned up four more things the exclusion had been hiding, all fixed here: **WH001 has a blind spot** — `no-hard-wrapped-prose.mjs` classifies any line indented four or more spaces as an indented code block, so a *nested* list item is never joined, which left three hard-wrapped bullets in `.github/workflows/README.md` §"Adding a job" that the autofix could not see (unwrapped by hand; they were the last hard-wrapped prose paragraphs in the repo) and made `AGENTS.md`'s and `development.md`'s "a list item is joined as a unit" wrong for nested items (both now state the four-space caveat); the `scripts/docs-prose.sh` header told readers to keep its denylist in lockstep with **two** sibling copies when there are **three** — the missed one being `.claude/agents/docs-reviewer.md`, the gating subagent's own system prompt, which had in fact been silently out of sync for the whole life of the `posthog-setup-report.md` exclusion; the `.markdownlint-cli2.jsonc` header's "applies to every tracked Markdown file" was exact for WH001 but not WH002, which returns early on anything that isn't `.mdx`; and the job-graph diagram omitted `docs-deploy`'s `needs` edges from `unit`, `integration`, and `e2e`, contradicting invariant 2 three lines below it. The denylist also drops its `PERF-CLAIMS-REVIEW.md` entry — unlike the wizard artifact this one names a file that was **never tracked** at all, so it guarded a hypothetical; the list's other general cases are patterns (`*.draft.md`, `*.old.md`) that already cover a one-off review document, and a literal filename restated in four places is the outlier. `scripts/docs-prose.sh all` still resolves the same 27-file prose set. ### Removed +- **`policy.LiteralValue` and `policy.CanonicalNumericLiteral`** (`internal/policy/{canonical,policy}.go`): the marker type and the numeric re-reading of a policy-authored insert-check literal. Insert checks are chtypes filters now, so a literal binds as written and ClickHouse reads it under the column's type — there is no second, numeric reading at compare time. A `_eq: "1.0"` against a `UInt64` used to admit a stored `1`; it is now ClickHouse's code 53 `TYPE_MISMATCH` per row (`422`), and the fix is to write a literal the column can read. `CanonicalScalar` stays: it is still the one rendering layer for a JWT claim. The released-version entry further down this file describing `LiteralValue` as shipped behaviour is left as history. + - **The policy's entire HTTP surface — `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, and the SDK's `wh.policy` namespace** (`internal/api/policy.go` + `policy_test.go` (deleted), `internal/api/{router,router_test}.go`, `cmd/wavehouse/main.go`, `clients/ts/src/policy.ts` (deleted), `clients/ts/src/{client,types,index}.ts`, `tests/e2e/sdk/{admin,query,ingest,streaming}.test.ts` + `settings.ts`, `docs/src/content/docs/{api.md,access-control.mdx,settings-directory.mdx,architecture.md,configuration.mdx,development.md,reverse-proxy.mdx,sdk/admin.md,sdk/reference.md}`, `AGENTS.md`; closes [#514](https://github.com/Wave-RF/WaveHouse/issues/514)): both endpoints were born alongside `PUT /v1/ops/policy` and outlived it when [#508](https://github.com/Wave-RF/WaveHouse/pull/508) deleted the policy write API; with files as the only write path, the policy is read, edited, and validated where it lives, so the whole read/dry-run surface goes too. The dry run had also kept its original lenient decoder while adoption became strict, certifying `{"valid": true}` for documents a reload would refuse — a misspelled operator key (`"eq"` for `"_eq"`) silently dropped into a filter that disables row security, the exact fail-open [#460](https://github.com/Wave-RF/WaveHouse/issues/460) demonstrated; deleting it removes the last non-strict policy decode site, closing #514 (the other five sites it cites were deleted or made strict by #508). Its replacement is `wavehouse validate`, which enforces strictly more (the cross-file role references against `roles.json` were invisible to a single-document dry run). The break-glass story narrows accordingly: the operator key can still trigger `POST /v1/ops/settings/reload`, whose findings report exactly why a rejected directory was refused — what it can no longer do is read back the adopted snapshot over HTTP; a bad edit still never breaks a running server (the previous good snapshot stays adopted). The e2e suite's read-modify-write helper pattern moves from `wh.policy.get()` to reading the harness-owned `policies.json` directly (`readPolicyFile()` — the file *is* the adopted policy there, since `setPolicy` fails unless the reload reports adoption). The policy document types (`Policy`, `TablePolicy`, `RolePermissions`) stay exported from the SDK — they describe `policies.json` and the e2e harness consumes them; `GET /v1/ops/pipes[/{name}]` is untouched. - **The README coverage badge and its whole publishing pipeline** (`.github/workflows/ci.yml`, `scripts/ci/publish-badge.sh` (deleted), `scripts/cov/main.go`, `.github/workflows/README.md`, `.testcoverage.yml`, `AGENTS.md`): [#502](https://github.com/Wave-RF/WaveHouse/pull/502) rewrote the README badge row and dropped the Go Coverage badge, but nothing removed what fed it — so for the six days until this landed the non-gating `badge` job kept running on every main push, holding `ci.yml`'s only `contents: write`, publishing `coverage-go.json` to the orphan `badges` branch for a badge no page rendered. Retired rather than restored ([#509](https://github.com/Wave-RF/WaveHouse/issues/509)): the `badge` job, its two producer steps in `coverage` (`cov badge` + the `go-coverage-badge` artifact), `scripts/ci/publish-badge.sh`, and the `cov badge` subcommand (with `badgeData`/`badgeColor`). The orphan `badges` branch is deleted separately once this lands — while the job still exists on main, the next code push would recreate it. **The gate is untouched** — `make cov`, `.testcoverage.yml`'s `threshold.total` and per-suite minima, and the GitHub Code Quality PR comments (the other half of [#133](https://github.com/Wave-RF/WaveHouse/issues/133)) all still run; only the published badge surface is gone. The security consequence is the reason to prefer retiring over restoring: **`ci.yml` now declares no `contents: write` in any job**, so the workflow that executes PR-authored code can no longer write to the repository under any path. Also fixed in passing: `timing`'s `needs` still listed `badge` (a dangling `needs` is a workflow-level error once the job is gone), and two permission comments were wrong: the "sole holder of `contents:write`" claims were repo-wide statements only ever true within `ci.yml` (`release.yml` and `publish-npm.yml` hold it too), and the workflow header claimed `docs-preview` was the only job with a write scope, undercounting `coverage`'s `code-quality: write` — which it holds while executing the PR tree. @@ -115,6 +129,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The `?token=` strip no longer repairs a query string that does not parse** (`internal/auth/auth.go`): `bearerToken` removed a query-string token by parsing the query, deleting `token`, and re-encoding what was left — and `url.ParseQuery` skips a pair it cannot read, so the re-encoding erased that pair. A handler that parses the query strictly in order to refuse a malformed one would then see a clean query: `GET /v1/ops/pipes?tenant=acme;x=1&token=…` would have answered `200` with the default tenant's pipes. The token is read exactly as before and a query that parses is rewritten exactly as before; a query that does not parse now loses its token pairs and nothing else, byte for byte. Pinned through `api.NewRouter` with the real authenticator, since a handler-level test never runs the middleware that rewrote the URL. - **SSE gap-fill re-reads the policy per replayed row** (`internal/stream/hub.go`): `ReplayProjector` captured the policy once when the replay began, so a policy adopted mid-fill — a revoked grant, say — applied only after the fill ended. It now reads it per event, as `Broadcast` does on the live path. +- **A row-filter claim containing a backslash, tab or newline no longer withholds rows the query path returns** (`internal/chsql/chsql.go`, `internal/chsql/chsql_test.go`, `internal/query/builder.go`, `internal/typelayer/filter.go`, `internal/typelayer/filter_test.go`, `tests/integration/rowfilter_stream_test.go`): ClickHouse reads a scalar `{p:String}` parameter with its escaped-text reader, so an unescaped `a\b` arrived holding a backspace and a raw tab or newline was a hard parse error. The stream compared such a claim false and the query path could misread it. Both surfaces now encode `\`, tab, newline and carriage return through one `chsql.EscapeStringParam`, measured byte-for-byte against ClickHouse 26.6.3.62 and against the chtypes artifact. The old behaviour was fail-closed — rows withheld, never leaked — so this is availability, not confidentiality. - **`classify-paths.sh` no longer reads a `grep` failure as "no match"** (`scripts/classify-paths.sh`, `scripts/classify-paths.test.sh`): both decisions were `if printf … | grep -qE …; then A; else B; fi`. `grep` exits `0` on match, `1` on no match and **`2` on error** (can't fork/exec, read error, bad pattern), and the `else` branch collapsed `1` and `2` into the same answer — `set -euo pipefail` does not help, since `set -e` is suppressed for a command used as an `if` condition. Observed twice in local `make ci` runs whose static checks run at `-j 14`: a different single case failed each time (`mixed-docs-go` answering `docs=false`, then `dep-bump-go` answering `code=false`) while every other case passed, which is the signature of a transient `grep` failure rather than a pattern bug. The test caught it only because it asserts expected values; **the production path has no such check** — CI's `changes` job gates the docs pipeline on this answer, so a `docs=false` produced by an errored `grep` silently skips the docs build and still reports success. The two greps now go through a `matches` helper that aborts with a diagnostic on any exit above 1, and the test suite stubs `grep` onto `PATH` to prove the abort fires (that case fails against the previous script). A second instance of the same class, found reviewing the first fix: the helper piped its input into `grep -q`, which exits at the first match — so once the file list outgrew the pipe buffer (a few thousand paths) the upstream `printf` died of SIGPIPE, `pipefail` reported 141, and the new error arm aborted on an ordinary large change set. Reproduced at 5,000 paths. It now reads from a here-string instead, and the suite pins that case. `scripts/ci/classify-changes.sh` also stopped reading the classifier through process substitution, which discarded its exit status: a classifier that aborted left `code`/`docs` empty, every `needs.changes.outputs.code == 'true'` job skipped, and the `CI` aggregator reported green having run nothing. It now captures the status, and fails closed — running everything — on a failed *or* partial classification, matching the rule already used for an empty file list. Also here, unrelated and one line: `biome.json` declared `$schema` 2.4.15 while the lockfile pins the 2.5.8 CLI, so `biome check --error-on-warnings` failed on the config itself for any change touching TypeScript. Bumped to match; it changes no lint rule. @@ -129,6 +144,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The pipes page no longer says a parameter can never break out of its literal** (`docs/src/content/docs/pipes.mdx`, `internal/pipes/pipes.go`): that holds only for a placeholder written bare. A string value brings its own quotes, so inside a quoted placeholder they close the template's: the body `{"id": " OR 1=1 OR id = "}` turns `WHERE id = '{{id}}'` into `WHERE id = '' OR 1=1 OR id = ''`, which matches every row. The page now says to write each placeholder bare, never inside quotes. Check existing `pipes.json` templates for quoted placeholders (`'{{x}}'`) and write them bare ([#662](https://github.com/Wave-RF/WaveHouse/issues/662)). - **An empty HMAC secret no longer verifies tokens signed with an empty key** (`internal/auth/auth.go` (+ tests), `SECURITY.md`): with `auth.jwt_secret` unset and no `auth.jwks_url` — the documented public-access posture, "no token can validate" — the key function handed `golang-jwt` an empty HMAC key, and the library verifies a token signed with one, so anyone could mint `{"role": "admin"}` and reach the whole data plane and `/v1/ops/*`. The verifier now refuses every token when it has neither a secret nor a JWKS URL, pinned by a test that signs with the empty key. Found by review on [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9 ([#604](https://github.com/Wave-RF/WaveHouse/pull/604)), which carries the same fix. +- **A policy claim compared against an integer column goes through a strict cast, so a claim that does not fit the column matches nothing instead of wrapping** (`internal/chsql/chsql.go`, `internal/policy/policy.go`, `internal/query/builder.go`, `internal/typelayer/{filter,typelayer}.go`, `docs/src/content/docs/{access-control.mdx,api.md,architecture.md}`, + tests): a claim bound as a plain `{p:String}` wrapped modulo 2^64 on every integer column (and at their own width on `[U]Int128`/`[U]Int256`), identically on `/v1/query`, the stream and the insert check, so a claim of `18446744073709551621` read and wrote tenant `5`. Row filters and insert checks now compare an integer column (`Nullable`/`LowCardinality` included) against `if(toString(accurateCastOrNull({p:String}, 'T')) = {p:String}, accurateCastOrNull({p:String}, 'T'), NULL)`: an in-range canonical claim answers exactly as before and keeps the primary key in use, while an out-of-range or non-canonical one (`007`, `+5`, `1.0`) matches no row on any operator and an insert check refuses it with `403` (a non-canonical claim used to be a per-row code 53 — no rows on a read, `422` on an insert). Other column types and a caller's own filters keep the plain form. - **Policy validation now rejects the fail-open rule shapes strict decoding can't see** (`internal/policy/policy.go`, `docs/src/content/docs/access-control.mdx`; closes [#460](https://github.com/Wave-RF/WaveHouse/issues/460)): four new `validateRolePerms` rejections close the fail-open shapes strict decoding can't see because the document is syntactically innocent. A `filter` entry with no operator (`"tenant_id": {}`) resolved to zero predicates — no `WHERE` clause, row security silently off, the same shape a misspelled `"eq"` for `"_eq"` used to decode to before strict decoding closed that route; it is now rejected, as is its check-path twin (an operator-less `check` entry, skipped by `Evaluate`'s resolve switch — accepted but constraining nothing) and `filter:` under an `insert:` grant (resolved and then ignored by the ingest path — the same accept-but-ignore family as [#224](https://github.com/Wave-RF/WaveHouse/issues/224), and the pointed asymmetry #460 called out against the loud `check` `_neq`/`_gt`/`_lt` rejection) along with its mirror, `check:` under a `select:` grant — the likelier authoring slip and the fail-open direction: the author believes reads are row-scoped while `Evaluate` resolves the entry and nothing on the select or stream paths reads it. Because [#508](https://github.com/Wave-RF/WaveHouse/pull/508) funneled every adoption through the one `policy.Validate` path, the four checks land on boot, the directory watch, `SIGHUP`, `POST /v1/ops/settings/reload`, and `wavehouse validate` at once. #460's migration caveat (a stored policy hard-failing at boot) has evaporated with the settings directory being new and unreleased; no shipped seed, compose, or fixture policy carries any of the rejected shapes. diff --git a/README.md b/README.md index 02582283..e317a429 100644 --- a/README.md +++ b/README.md @@ -14,7 +14,7 @@

The open-source real-time API gateway for ClickHouse: schema-aware ingest, async batching, real-time SSE streaming, and tiered query caching. - All in a single binary. + All in one binary, plus the per-ClickHouse-version artifact it loads at start.

@@ -67,7 +67,7 @@ Full walkthrough at **[wavehouse.dev/getting-started](https://wavehouse.dev/gett ## Why WaveHouse? -ClickHouse is a phenomenal OLAP database, but pointing a frontend right at it leaves a lot to be desired: one-row inserts trigger `Too many parts`, there's no backpressure or edge validation, no real-time push, and no row/column security. You end up building custom APIs, a Kafka queue, a batch consumer, a cache tier, and an auth service. **WaveHouse is that whole stack as one binary** — the only external dependency is ClickHouse. +ClickHouse is a phenomenal OLAP database, but pointing a frontend right at it leaves a lot to be desired: one-row inserts trigger `Too many parts`, there's no backpressure or edge validation, no real-time push, and no row/column security. You end up building custom APIs, a Kafka queue, a batch consumer, a cache tier, and an auth service. **WaveHouse is that whole stack as one binary** (plus a per-ClickHouse-version artifact it loads at start, for ClickHouse-native ingest validation and row-level security) — the only external network dependency is ClickHouse. If you're building user-facing analytics, WaveHouse is like **Supabase for ClickHouse**. Or an **open-source Tinybird** that pushes data to the frontend in real time over SSE, not just pull-based REST. @@ -81,7 +81,7 @@ If you're building user-facing analytics, WaveHouse is like **Supabase for Click | | Direct ClickHouse | Kafka + CH (DIY) | Tinybird | **WaveHouse** | | ----------------------------- | :---------------: | :--------------: | :-----------: | :------------: | -| Self-hosted, single binary | — | — | ✗ (SaaS) | ✓ | +| Self-hosted, one binary | — | — | ✗ (SaaS) | ✓ | | Safe high-rate inserts | ✗ | ✓ (via Kafka) | ✓ | ✓ | | Schema validation at the edge | ✗ | custom | ✓ | ✓ | | Real-time push (SSE) | ✗ | custom service | ✗ | ✓ native | @@ -127,6 +127,14 @@ Swap in `:vX.Y.Z` and `release.yml` for a release image. Pin the signer either w go install github.com/Wave-RF/WaveHouse/cmd/wavehouse@latest ``` +`go install` compiles from source with cgo enabled (requires a C toolchain and glibc — Linux amd64/arm64 or macOS arm64) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse loads at start. Fetch it once before the first run: + +```bash +go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch +``` + +This downloads 160–290 MB into the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision); point `WH_CHTYPES_REGISTRY` elsewhere if you keep it somewhere else. + ```bash wavehouse bootstrap ./settings # starter settings directory, every key at its default WH_SETTINGS_DIR=./settings wavehouse @@ -144,7 +152,7 @@ Track what's shipped, in progress, and planned on the [**project board**](https: ## Local Development -You'll need **Go 1.26+, GNU Make 4+, Docker (Compose v2), Node.js 22 LTS, and pnpm 11.21+**. See [development docs](https://wavehouse.dev/development) for the authoritative source of truth with the full list, version requirements, and gotchas. +You'll need **Go 1.27+, GNU Make 4+, Docker (Compose v2), Node.js 22 LTS, and pnpm 11.21+**. See [development docs](https://wavehouse.dev/development) for the authoritative source of truth with the full list, version requirements, and gotchas. ```bash make tools # one-time bootstrap diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index c26e1917..b82f0ce7 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -190,11 +190,13 @@ The rules, in order: 2. **An empty (or `["*"]`) `allow_columns` means "all columns"** — every column not in `deny_columns` is permitted. Use this with `deny_columns` for a blocklist posture: see everything *except* a few sensitive columns. 3. **A non-empty `allow_columns` is an allowlist** — only the named columns (and never the denied ones) are permitted. -On a structured query (`POST /v1/query?table={table}`) the allowlist is a **hard cap on every column the query references — in any clause**: the projection, an aggregation argument, `filters`, `group_by`, `order_by`, and `time_range`. Naming a disallowed column anywhere is rejected with `403 column "x" not allowed`. A full-row read is requested explicitly with `"select_all": true`, which expands to exactly the columns the role may read — never a raw `SELECT *` that could include a denied column; if the role is allowed *no* columns, the read is rejected (`403`) rather than returning empty rows. **Omitting `columns` (or sending `[]` / `""`) returns nothing** — a request for no data — so a hidden column can't leak by being left out, grouped on, or filtered on to infer its values. (Note: in a query, `["*"]` is the *literal column named `*`*, not a wildcard — use `select_all` for all columns. In `allow_columns`, `["*"]` is still the all-columns wildcard.) On insert (`POST /v1/ingest?table={table}`) the body is `403 column "x" not allowed for insert`. On live streams, denied columns are silently **stripped** from each event rather than rejecting the connection. The structured-query and live-stream paths defer to the **same** per-column decision (`IsColumnAllowed`), so the two read surfaces enforce identical column visibility and can't drift apart. +On a structured query (`POST /v1/query?table={table}`) the allowlist is a **hard cap on every column the query references — in any clause**: the projection, an aggregation argument, `filters`, `group_by`, `order_by`, and `time_range`. Naming a disallowed column anywhere is rejected with `403 column "x" not allowed`. A full-row read is requested explicitly with `"select_all": true`, which expands to exactly the columns the role may read — never a raw `SELECT *` that could include a denied column; if the role is allowed *no* columns, the read is rejected (`403`) rather than returning empty rows. **Omitting `columns` (or sending `[]` / `""`) returns nothing** — a request for no data — so a hidden column can't leak by being left out, grouped on, or filtered on to infer its values. (Note: in a query, `["*"]` is the *literal column named `*`*, not a wildcard — use `select_all` for all columns. In `allow_columns`, `["*"]` is still the all-columns wildcard.) On insert (`POST /v1/ingest?table={table}`) the rule is enforced by compiling the role its *own* copy of the table schema, without the columns it may not write — so a record naming one is refused by ClickHouse's parser with code **117**, `Unknown field found while parsing JSONEachRow format: x`, the same answer a column the table does not have gets. The message no longer confirms whether the column exists. On live streams, denied columns are silently **stripped** from each event rather than rejecting the connection. The structured-query and live-stream paths defer to the **same** per-column decision (`IsColumnAllowed`), so the two read surfaces enforce identical column visibility and can't drift apart. + +On ingest (`POST /v1/ingest?table={table}`) the same lists cap what a role may write, and a column it may not write is indistinguishable from one the table does not have: the record is refused by ClickHouse's own parser with code `117` (`Unknown field found while parsing …`), a `400` — per record in a batch, or for the whole request when the body is a single object or has a `header=present` header naming it. There is no separate `403` for a denied column on insert, and the published row never carries one. ## Row-level security -`filter` restricts *which rows* a role can read. Like the per-column decision above, one resolution drives **both** read surfaces: a structured query gets the predicates injected as a `WHERE` clause in the generated SQL, and the live stream evaluates the *same* resolved predicates in memory against each subscriber's claims before delivering an event — so row visibility can't drift between the two paths (the stream's in-memory comparison has a fail-closed boundary; see [the enforcement caution below](#where-each-rule-is-enforced)). Each entry maps a column to a comparison whose value is usually a **JWT claim template**: +`filter` restricts *which rows* a role can read. One resolution drives **both** read surfaces: the structured query injects the predicates as a `WHERE` clause, and the live stream compiles the *same* resolved predicates through chtypes and evaluates them per subscriber before delivering an event — so row visibility cannot drift between the two paths (fail-closed boundary; see [the enforcement caution below](#where-each-rule-is-enforced)). Each entry maps a column to a comparison whose value is usually a **JWT claim template**: ```json { @@ -239,13 +241,25 @@ Any `filter` or `check` operator value may interpolate token claims with `{{ jwt - `{{ jwt.sub }}` → the token's `sub` claim. - `{{ jwt.app_metadata.tenant_id }}` → a nested claim. -Values are always bound as SQL **parameters**, never concatenated into the query, so templating is injection-safe. If a claim path in a `filter` template can't be resolved (a validly-signed token that simply doesn't carry the claim), that filter **fails closed**: the predicate becomes constant-false, so on the structured-query path (`POST /v1/query`) the role sees **no rows**, and the live stream withholds every event for that subscriber — one resolution drives both read surfaces (see [where each rule is enforced](#where-each-rule-is-enforced)). This holds for every operator — `_eq`, `_neq`, `_gt`, `_lt`, and `_in` alike. The alternative, binding the empty string the template would render to, would leave a live predicate against `''`: `_eq` would match every empty-valued row, and `_neq`/`_gt` on a string column would match essentially *all* rows, erasing the restriction. A literal value with no template in it — including an explicit `""` — binds exactly as written, even when it spells a number ([insert checks](#insert-checks) additionally accept such a literal's numeric reading at compare time, since a policy literal carries no JSON type — but what *binds* is always the spelling you wrote; whether it then matches follows the column's type: a `String` or byte-equality column compares that exact text, while a numeric column reads the constant as a number on both surfaces — on a `Float` or `Decimal` column `_eq: "1.0"` admits a stored or streamed `1`, but an integer column accepts only the plain digit form: `1.0` and `1e3` alike are refused, matching the cast error the query path raises for them (see [the enforcement caution](#where-each-rule-is-enforced))). +Values are always bound as SQL **parameters**, never concatenated into the query, so templating is injection-safe. If a claim path in a `filter` template can't be resolved (a validly-signed token that simply doesn't carry the claim), that filter **fails closed**: the predicate becomes constant-false, so on the structured-query path (`POST /v1/query`) the role sees **no rows**, and the live stream withholds every event for that subscriber — one resolution drives both read surfaces (see [where each rule is enforced](#where-each-rule-is-enforced)). This holds for every operator — `_eq`, `_neq`, `_gt`, `_lt`, and `_in` alike. The alternative, binding the empty string the template would render to, would leave a live predicate against `''`: `_eq` would match every empty-valued row, and `_neq`/`_gt` on a string column would match essentially *all* rows, erasing the restriction. A literal value with no template in it — including an explicit `""` — binds exactly as written, and ClickHouse reads it under the column's type: a `String` column compares that exact text, while a numeric column reads the constant as a number. A spelling the column cannot read is not a second chance — `_eq: "1.0"` against a `UInt64` matches no rows on a read filter and is refused with `403` on an insert (an integer column takes only the canonical spelling; see the caution below), and on another numeric column such as `Decimal` it is ClickHouse's own code 53 `TYPE_MISMATCH` per row, which withholds on a read filter and is a `422` on an insert. Write a literal the column can read. + +"Can't be resolved" means the claim path is **absent** from the token (or `null`) — or resolves to a JSON **object or array** rather than a scalar, which usually means a dropped path segment (`{{ jwt.app_metadata }}` where `{{ jwt.app_metadata.tenant_id }}` was meant); the one structured shape with defined semantics is the bare-claim `_in` array above. Scalar claims — strings, booleans, and numbers — resolve normally. A numeric claim binds in **canonical decimal form**, not the token's spelling: `1.0` and `1e3` bind as `1` and `1000`, so spelling differences between issuers never change the bound value, and an integer id keeps every digit up to a ~100-digit bound on the value's *exact* decimal form (so `1e150` is refused even though a float64 holds it — prefer issuing integer ids as integers). Type the scoped column to match: an integer column, or a `Decimal` whose scale covers the claim's fractional digits, keeps the comparison exact for ids that fit the column's type, while a `Float32`/`Float64` column rounds the stored value and gives that exactness back (`col = '9007199254740993'` matches a stored `9007199254740992` there). A claim that is *present but empty* is a value the token vouches for: it resolves to `''` and binds normally. Make sure your identity provider issues the claims your policy templates reference — and omits unset claims rather than issuing them as empty strings. + +:::caution[Choosing columns to compare claims against] +Every claim is bound as a string parameter and read under its column's type — on `/v1/query` and on the stream alike, and for an auto-injected `_eq` value too — so the column's type decides how the claim is read. + +Compare claims against `String` or `UUID` columns (tenant ids, org ids, user ids): those compare the claim exactly as written. + +On an integer column (`UInt8` through `UInt256`, `Int8` through `Int256`, `Nullable` included) the claim must still fit the column's type, and is compared through a strict cast: a claim that is not the canonical spelling of a value the column can hold — out of range such as `18446744073709551616` (2^64), or spelled `007` or `+5` — matches no rows on any operator, and an insert `check` refuses the record with `403`. It never wraps onto another value. -"Can't be resolved" means the claim path is **absent** from the token (or `null`) — or resolves to a JSON **object or array** rather than a scalar, which usually means a dropped path segment (`{{ jwt.app_metadata }}` where `{{ jwt.app_metadata.tenant_id }}` was meant); the one structured shape with defined semantics is the bare-claim `_in` array above. Scalar claims — strings, booleans, and numbers — resolve normally. A numeric claim binds in **canonical decimal form**, not the token's spelling: an integer id keeps every digit — up to a 100-digit bound, far past any real id — while `1.0` or `1e3` binds as `1` and `1000`, so spelling differences between issuers never change the bound value. What must fit the bound is the value's **exact decimal form**, exponent applied — roughly 100 digits (the exact-form gate allows 102 characters, whatever they are) — so `1e400` can't be resolved, but neither can `1e150` or `1e-150`, whose short spellings expand to 151- and 152-character exact forms even though a float64 could hold them; prefer issuing integer ids as integers. Type the scoped column to match: an integer column — or a `Decimal` whose scale covers the claim's fractional digits — keeps the comparison exact, while a `Float32`/`Float64` column rounds the stored value and quietly gives that exactness back (`col = '9007199254740993'` matches a stored `9007199254740992` there). A claim that is *present but empty* is a value the token vouches for: it resolves to `''` and binds normally, so `_neq` against an empty-string claim still emits `col != ''`. Make sure your identity provider actually issues the claims your policy templates reference — and omits unset claims rather than issuing them as empty strings. +On a timestamp column (`Date`, `DateTime`, `DateTime64`) the claim is parsed the way ClickHouse parses a string compared with that type. A claim without a UTC offset is read in the server's time zone, and one carrying an offset such as `2026-01-01T00:00:00Z` is accepted only on ClickHouse 26.5 and later — earlier lines refuse the comparison. + +Prefer identity columns over time columns, and express time windows in the query instead. +::: Claim paths may contain only letters, digits, `_`, and `.` (the segment separator). Any `{{ … }}` fragment that is not a well-formed `{{ jwt. }}` template — a path outside that grammar (a hyphen: `{{ jwt.tenant-id }}`; a namespaced claim: `{{ jwt.https://app.example.com/tenant_id }}`), a missing or misspelled `jwt.` prefix, or an unterminated `{{` — is **not** recognized as a template, and left unchecked the resolver would bind the literal `{{ … }}` text as a value. (Policy values have no other placeholder syntax; a pipe's `{{param}}` placeholders are a different mechanism and are not accepted here.) That is *not* fail-closed: on a read filter `_neq`/`_lt` would then match essentially every row (a leak), and on a write `check` the literal text would be stamped into every inserted row (silent corruption). So such a policy is **rejected when it is validated**: a `policies.json` carrying one refuses boot, is refused by a reload (the previous good policy stays in effect), and fails `wavehouse validate`. Every adoption runs the current rules, so an upgrade that tightens validation re-checks the file on the next boot or reload. Flatten hyphenated or namespaced claims into a supported path at your identity provider — on Auth0, an [Action can set a flat custom claim](https://auth0.com/docs/secure/tokens/json-web-tokens/create-custom-claims) outside the registered OIDC names (its legacy Rules required URL namespacing, which established tenants often still carry), and namespacing is a common OIDC convention elsewhere, so a namespaced claim is the shape you are most likely to meet first. -Row filters apply on the structured-query path and, per subscriber, on the live SSE stream — the stream evaluates the same resolved predicates in memory against each subscriber's token claims (see the [enforcement caution](#where-each-rule-is-enforced) for what its in-memory comparison can and cannot decide). Named pipes authorize by `allowed_roles` membership alone — scope a pipe's exposure in its SQL text, since neither the table policy's row `filter` nor its column allow/deny list is applied on the pipe path (see [Named Pipes](/pipes#authorizing-a-pipe)). +Row filters apply on the structured-query path and, per subscriber, on the live SSE stream — the stream compiles the same resolved predicates through the in-process ClickHouse parser and evaluates them against each subscriber's token claims (see the [enforcement caution](#where-each-rule-is-enforced) for the fail-closed reasons a compile or evaluation can withhold a row). Named pipes authorize by `allowed_roles` membership alone — scope a pipe's exposure in its SQL text, since neither the table policy's row `filter` nor its column allow/deny list is applied on the pipe path (see [Named Pipes](/pipes#authorizing-a-pipe)). ## Insert checks @@ -269,13 +283,14 @@ Row filters apply on the structured-query path and, per subscriber, on the live } ``` -For each checked column, on `POST /v1/ingest?table={table}`: +Checks are compiled into **one chtypes filter** — `col = {p:String}` for `_eq`, `col IN (…)` for `_in`, AND-joined over every checked column, with an integer column's claim compared through the strict cast described under [JWT claim templating](#jwt-claim-templating) — and evaluated against the row ClickHouse's own parser produced, by the same engine that evaluates a row `filter`. So a check sees the **stored** value, after coercion and after `DEFAULT`s: what it admits is what the table will hold. -- **If the request body includes the column**, its value must satisfy the check — equal the claim-derived value (`_eq`), or be one of the claim-derived set (`_in`) — or the insert is rejected with `403 check failed for column "x"`. Equality is decided on **canonical scalar form**, the same rule the claim side follows: a numeric body value matches by value, not spelling (`1.0` satisfies a check against claim `1`); a body value with no canonical form — an object, array, or `null` — satisfies no check, so a `check` on a `Map` or `Array` column rejects every insert that names it; and the 100-digit bound applies to the body's literal too, so an over-long numeric value is this same `403`, not a schema error. A **template-free** `_eq` check value carries no JSON type, so it matches by either reading — a static `_eq: "1.0"` accepts an inserted `1.0` (number) and an inserted `"1.0"` (string) alike, while auto-injecting exactly as written. A claim-derived value keeps strict canonical equality: a string-typed claim matches only its exact text, so a claim of `"1e3"` never accepts an inserted `1000`. On an integer or `Decimal` column a `writer` therefore cannot forge a row for another user or tenant — equal canonical forms store equal values. On a `Float32`/`Float64` column the rounding [noted above](#jwt-claim-templating) applies to the write side too: the stored value can land on a neighboring id. And on a `String` column the guarantee is narrower in a different way: the check compares canonical forms but ClickHouse stores the body's raw spelling, so a body value that is an alternate numeric spelling of the claim (`1e3` for a claim of `1000`) passes the check yet stores differently than the auto-injected value would — and a literal's numeric reading shares this property (`_eq: "1.0"` accepting an inserted `1` stores the text `1`, not `1.0`). -- **If the request body omits the column**, an `_eq` check **auto-injects** the claim-derived value before publishing, so clients can send just the business fields (`page`, `score`) and let the policy stamp `user_id` and `tenant_id` from the token. An `_in` check has no single value to stamp, so an omitted column is **rejected** — the writer must name a value within their allowed set. -- **The check's column must be one a record can actually carry.** A `check` naming a column the table does not have, one ClickHouse computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is refused with a per-record `403` naming the column — on *every* insert by that role, until the policy or the table is corrected. None of the three can be enforced: the published row has one slot per insertable column, so an injected value for a computed or unknown column is dropped on the way out, and an ephemeral column is accepted by the `INSERT` but never stored and never selectable. Each would have answered `200` while enforcing nothing, which is worse than refusing. `wavehouse validate` cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**. (The `EPHEMERAL` case is refused even when the body *supplies* the value: it passes schema validation, since an ephemeral column is legal in an `INSERT`.) +- **If the request body includes the column**, its value must satisfy the check or the record is rejected with `403 check failed for column "x"` (`… for columns "x", "y"` when more than one is checked — the filter is AND-joined, so it names the set tested rather than inventing an attribution). Comparison is ClickHouse's, under the column's own type: on an integer, `Decimal`, `UUID` or `String` column a writer cannot forge a row for another tenant, because equal values store equal. A `Float32`/`Float64` column rounds the stored value and gives that exactness back — the row can land on a neighboring id. A check the engine cannot evaluate at all is a `422`, never a silent pass. +- **If the request body omits the column** — or sends an explicit `null`, which `input_format_null_as_default` resolves the same way — an `_eq` check **auto-injects** the claim-derived value, so clients can send just the business fields and let the policy stamp `user_id` and `tenant_id` from the token. It is implemented as a `DEFAULT` on the role's compiled schema, so a value the caller *does* send still wins. An `_in` check has no single value to stamp, so the **table's own default** is what gets tested: the record is admitted if that default is in the claim-derived set and rejected if it is not. +- **The check's column must be one a record can actually carry.** A `check` naming a column the table does not have, one ClickHouse computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is refused with a per-record `403` naming the column — on *every* insert by that role, until the policy or the table is corrected. None of the three can be enforced: the published row has one slot per wire column, so an injected value for a computed or unknown column is dropped on the way out, and an ephemeral column is never stored. Each would have answered `200` while enforcing nothing. `wavehouse validate` cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**. +- **A claim literal the column cannot read fails closed, not loosely.** A `_eq` value that is not a legal literal for the column (`"1.0"` on a `UInt64`) cannot be compiled as that column's `DEFAULT`, so the injection is dropped and logged; the filter then judges the record as sent, which an absent column loses. On an integer column every record then fails the check — a `403`; on another column a supplied value is ClickHouse's code 53 `TYPE_MISMATCH` per row — a `422`. -**When the claim can't be resolved** (a validly-signed token that doesn't carry it), an `_eq` check is — unlike a row filter — **not** fail-closed: the template still renders, the unresolvable placeholder replaced by the empty string and any surrounding literal text kept (`"acct-{{ jwt.org_id }}"` → `acct-`, a bare `"{{ jwt.sub }}"` → `''`), and that rendered value becomes the required value — so an omitted column is auto-injected with it and any other supplied value is rejected. On an integer or `Decimal` column that stamped `''` does not stay empty: ClickHouse coerces it to `0`, so every such row lands on tenant/user `0` — and since the read side of the same claim fails closed, those rows are also invisible to the writer that produced them. (A `Float` column may instead reject the coercion — ClickHouse-version-dependent — failing the insert into the DLQ.) An unresolvable `_in` check instead fails closed: the claim-derived set is empty, so every insert by that role into that table is rejected (there is no single value to auto-inject). The `_in` element rule from [row-level security](#row-level-security) applies here too — one non-scalar element (`null`, object, nested array) in the claim array collapses the allowed set the same way. Whether the `_eq` write path should match the read path and reject the insert rather than stamp `''` is tracked in [#463](https://github.com/Wave-RF/WaveHouse/issues/463). +**When the claim can't be resolved** (a validly-signed token that doesn't carry it), an `_eq` check is — unlike a row filter — **not** fail-closed on a `String` column: the template still renders, the unresolvable placeholder replaced by the empty string and any surrounding literal text kept (`"acct-{{ jwt.org_id }}"` → `acct-`, a bare `"{{ jwt.sub }}"` → `''`), and that rendered value becomes the required value — so an omitted column is auto-injected with it and any other supplied value is rejected. On a **numeric** column it now fails closed: `''` is not a literal a `UInt64` can read, so the role's schema will not compile with it as a default and no record satisfies the check — every insert by that role into that table is refused (a `403` on an integer column, a `422` on a `Decimal` or `Float` one), rather than the rows landing on tenant/user `0` as they used to ([#463](https://github.com/Wave-RF/WaveHouse/issues/463)). An unresolvable `_in` check fails closed everywhere: the claim-derived set is empty, so nothing satisfies it. The `_in` element rule from [row-level security](#row-level-security) applies here too — one non-scalar element (`null`, object, nested array) in the claim array collapses the allowed set the same way. This pairs naturally with a matching `filter` on the `select` side: `check` stamps the tenant on write, `filter` scopes reads to that tenant. @@ -350,24 +365,14 @@ The same policy drives every data path, but not every field is meaningful on eve | Named pipe | `GET/POST /v1/pipes/{name}` | per-pipe `allowed_roles` (not the policy engine; see [Named Pipes](/pipes)). Resource limits come from ClickHouse's [server-wide settings](/configuration#server-side-resource-limits), not per-role policy caps | :::caution[Live streams enforce column and row policy, but not resource limits] -SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and — like the query path — receive only the rows their role's row-level `filter` predicates admit, evaluated per subscriber against their JWT claims. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role can't `select` (denied columns are still stripped from what's delivered). The stream evaluates predicates in memory — reproducing ClickHouse's coercion for the types it can classify and refusing the rest — so how much it can enforce depends on what the schema says about the filtered column, and every case it cannot decide **fails closed**. Provided the filter constant is one the query path's SQL also accepts (see the per-type guidance below — the *every other type* bucket constrains your constant to the event's own text rendering, and timestamp constants should be zone-less or Unix-seconds strings — the two spellings every ClickHouse release converts in a `WHERE` comparison), ambiguity only ever *withholds* a row the query path would return, never delivers one it would hide. The one residual payload-vs-stored case — an event whose insert later fails outright into the DLQ — is called out below. +SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -- **Numeric columns** (`Int*`/`UInt*`/`Float*`/`Decimal*`, unwrapping `Nullable`/`LowCardinality` in any nesting): all five operators compare numerically in the column's **storage domain**, matching ClickHouse. Both operands render to the claim side's exact [canonical decimal form](#jwt-claim-templating) — so a 64-bit ID never falsely matches a neighbor, string-encoded or bare — and are then narrowed the way ClickHouse narrows the stored value and the bound constant: `Float32`/`Float64` round to the column's width, `Decimal` truncates at its scale, and integer columns are exact at any width. An operand outside the JSON number grammar (`NaN`, any `Inf`/`Infinity` spelling, hex), past roughly the [100-digit bound](#jwt-claim-templating), or beyond the float domain's range withholds the row — as does, on an integer column, a fractional operand or a constant spelled any way but the plain form ClickHouse's integer cast accepts (`1e3` errors the query there, so the stream withholds to match). Operands outside the column's **numeric range** (a negative bound on an unsigned column, a value past the integer width, more integer digits than a Decimal's precision budget) are refused on both sides too: ClickHouse's own reading of such a constant varies by pair — an error, a mathematical promotion, or a width-boundary wrap onto a *different* value than written — so the stream withholds rather than model any one behavior, and an out-of-range payload was never storable anyway. Write bounds within the column's range. -- **`String` columns** (again under any `Nullable`/`LowCardinality` wrapping): byte comparison *is* ClickHouse's String comparison — equality and ordering are both exact. -- **`DateTime`/`DateTime64` columns** (again under any wrapping): both operands are parsed as **instants** — through the same grammar [ingest canonicalization](/api#timestamp-canonicalization) reads — and compared chronologically, so all five operators are exact and the constant's spelling doesn't need to match the event's: ingest rewrites payload values to RFC 3339 UTC before publishing, and a zone-less constant (`2026-06-21 04:00:00`, read in the column's declared zone, else the discovered server default — ClickHouse's own rule) still matches the rewritten payload denoting that instant. **Write timestamp constants zone-less like that, or as 9–10-digit Unix-seconds strings**: those two spellings are converted to the column type in a `WHERE` comparison on every ClickHouse release. The RFC 3339 `Z` form is instead rejected there with a type error on older releases (verified on 25.x; 26.6 accepts it — the exact release that changed isn't pinned here, so prefer the two always-safe spellings) — a property of the release's constant conversion, not of `date_time_input_format`, which governs the [ingest parser WaveHouse pins](/api#timestamp-canonicalization) rather than comparison constants, so setting `best_effort` server-side does not rescue it. The stream accepts every grammar spelling regardless. An operand the grammar can't read — on either side — withholds the row, as does an instant outside the column type's range (which insert-time *saturation* would have moved anyway). A column whose **declared** zone can't be loaded at runtime has no timestamp parser at all — it falls into the byte-equality bucket below (its values aren't canonicalized at ingest either), so `_neq`/`_gt`/`_lt` withhold every row. A column relying on the **server default** zone when that couldn't be resolved keeps instant comparison for zone-explicit operands (RFC 3339, Unix seconds) but refuses zone-less ones rather than guess the zone. -- **Every other type** (`Enum`, `UUID`, `Date`/`Date32`, `Bool`, `IPv4`/`IPv6`, `FixedString`, …): only byte-equality is trusted. `_eq`/`_in` admit exactly the event's own text rendering — write the filter value the way your events carry it (`true`, not `1`; a lowercase UUID if that's what clients send; `Date` values keep the producer's spelling — unlike `DateTime`, they are not canonicalized at ingest). The same constant is bound into the query path's SQL, where ClickHouse compares it against the column's declared type rather than the event's text rendering — so pick a value that works on both surfaces, and verify the query path returns what you expect before relying on the filter. `_neq`, `_gt` and `_lt` withhold **every** row: a text difference can be pure representation (an uppercase UUID, an alternate date format, an Enum name vs. its number), so inequality and order are unprovable without ClickHouse — on these columns, use the query path for ordering/exclusion filters. -- **No usable schema** — the table is unknown to schema discovery, or discovery is still failing at boot (the server serves while retrying in the background): every column is treated as the "other" bucket above (timestamp columns lose their parser too). Equality scoping keeps working; ordering and `_neq` withhold until a schema is available. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, ClickHouse's code 53; on an integer column such a claim is a `filter`), `decline` (the engine would not answer), `unavailable` (the tenant's ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list: a filter over a column the published row does not carry fails closed for that subscriber rather than being treated as a mismatch on an absent value. A filter naming a column the table no longer has, or a `MATERIALIZED`/`ALIAS`/`EPHEMERAL` one, fails to **compile**, which withholds every row for that role until the filter or the table changes — logged once, not per event. -A few more edges worth knowing when you write a policy — the stream evaluates the **ingested event payload**, not the stored row, and each of these follows from that: - -- **A filtered column the payload doesn't carry withholds *every* event** for that subscriber, even though the same filter matches normally on the query path. That bites a `MATERIALIZED`/`ALIAS` column, which is never part of an ingest payload — and note the `check` half of the pairing is no longer available for those columns, since a `check` on one is now refused outright (see Insert checks above). A `DEFAULT` column your clients omit is withheld too, but by the next rule rather than this one: a positional row carries one slot per insertable column, so an omitted column arrives as an explicit `null` rather than being absent. The recommended [`check` + `filter` pairing](#insert-checks) is unaffected: an `_eq` insert `check` auto-injects its claim value into any payload that omits the column *before* the event is published, so the streamed event carries it and the matching row filter evaluates normally. (That holds for timestamp columns too: the injected claim value is canonicalized with the rest of the payload before publish, and the stream compares timestamps as instants, so the claim's spelling and the canonical wire spelling meet.) -- **A non-scalar event value** (array/object/null) under a filtered column withholds the row. -- **Insert-time numeric narrowing is simulated, not skipped.** The insert narrows a payload carrying more precision than the column's declared type — a `Decimal` **truncates** at its scale (`1.005`, `1.006` and `1.009` all store as `1.00` in a `Decimal(10, 2)`), a `Float32` **rounds** to its nearest representable value (`16777217` stores as `16777216`) — and ClickHouse applies the same narrowing to a bound filter constant at compare time. The stream narrows **both operands** identically before comparing, so its verdict matches the query path's on narrowing columns under every operator: a `_gt: "1.004"` filter on a `Decimal(10, 2)` column withholds a `1.005` payload exactly as the query path hides the stored `1.00`. (An earlier revision of this feature compared the raw payload and could deliver such an event; that fail-open is closed, and an integration test holds every in-range numeric stream verdict equal to a live ClickHouse's — for the out-of-range operands the range gate refuses, it asserts the half that matters: the stream never admits a row ClickHouse hides.) What remains payload-vs-stored: an event whose insert later **fails outright** (a value ClickHouse rejects, such as one out of range, which the DLQ parks; a ClickHouse outage only delays the row, which is retried until it inserts) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. +Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. One more boundary is temporal: a subscriber's claims (and role) are captured when the SSE connection is established and are never re-read, while the policy itself is re-read on every event, live and replayed alike — so a policy adopted mid-gap-fill applies to the next replayed row. Tightening a policy therefore applies from the next event either way, but a token that expires — or claims revoked at the identity provider — keeps its open stream until the client disconnects, so treat connection lifetime as the revocation window for stream row-scoping. -Each row withheld **from a subscriber** by row-level security is counted in `wavehouse_sse_rows_withheld_total` (labeled by table and role; a row withheld from three subscribers counts three times) — check it before concluding a quiet stream simply has no matching rows. - The resource limits (`max_rows`, `max_execution_time`, `max_rows_to_read`, `max_memory_usage`) remain a property of the SQL query path and are **not** applied to the live event stream; if those caps are part of a role's isolation story, don't rely on them over the stream. ::: diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 1af6ee20..dd27e108 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -64,19 +64,12 @@ X-Content-Type-Options: nosniff The body is always a JSON object that includes an `error` field describing the failure: ```json -{"error": "invalid json"} +{"error": "unknown table: clicks"} ``` Some endpoints attach extra fields alongside `error` on their **failure** responses — e.g. a failing `/readyz` returns `{"status":"not ready","error":"…"}`. The guarantee is scoped to failures: whenever a response signals an error (any 4xx/5xx), an `error` field is present and parseable. Success responses carry each endpoint's own shape and need **not** include `error` — a healthy `/readyz` returns just `{"status":"ready"}`. -This contract holds for: - -- Handler-emitted errors — validation (4xx), permission denials (403), not-found (404), backend errors (5xx). -- Router-level **404 Not Found** when the URL does not match any registered route. -- Router-level **405 Method Not Allowed** when the URL matches a route but the method is not registered. -- Server-level **500 Internal Server Error** when a handler panics — recovered, logged with stack, and reported to the client as JSON **when the handler has not yet committed any response headers or body bytes**. - -Historically some error paths defaulted to `text/plain` because they were emitted via `http.Error` or chi's default handlers; those paths now route through a shared `writeJSONError` helper so strict clients can branch on `Content-Type` consistently. +The contract holds for handler-emitted errors (validation, permission denials, not-found, backend failures), for router-level `404`s and `405`s, and for the `500` a recovered handler panic produces — the last one only while no response headers or body bytes have been committed. Everything routes through one `writeJSONError` helper, so strict clients can branch on `Content-Type` consistently. The per-endpoint error tables below list the bodies you can expect for each status code; the `Content-Type` and `X-Content-Type-Options` headers above apply uniformly and are not repeated. @@ -160,7 +153,7 @@ Status code: `503 Service Unavailable` ### Liveness vs readiness — behavior matrix -`/livez` (liveness) and `/readyz` (readiness) answer different questions, so they diverge once the process has booted. `/livez` is **sticky**: after the first successful schema discovery it stays `200` for the rest of the process lifetime, even if ClickHouse later becomes unreachable — liveness asks "is the process alive and past boot," not "is its backend up right now." `/readyz` stays **conditional**: it pings ClickHouse on every call and drops back to `503` whenever ClickHouse is unreachable. +`/livez` is **sticky**: after the first successful schema discovery it stays `200` for the rest of the process lifetime, even if ClickHouse later becomes unreachable — liveness asks "is the process alive and past boot," not "is its backend up right now." `/readyz` stays **conditional**: it pings ClickHouse on every call and drops back to `503` whenever ClickHouse is unreachable. | State | `/livez` | `/readyz` | |----------------------------|:--------:|:---------:| @@ -192,7 +185,7 @@ Returns the build metadata embedded in the running binary — `version`, `git_co "version": "1.2.3", "git_commit": "a1b2c3d", "build_time": "2026-06-02T12:00:00Z", - "go_version": "go1.26.3" + "go_version": "go1.27.0" } ``` @@ -227,28 +220,30 @@ Every other route answers `404`, including every tenant route. Under `/v1/ops`, ### `POST /v1/ingest?table={table}` — Ingest Data -Accepts a single flat JSON object, a JSON array of objects, or a newline-delimited JSON (NDJSON) batch, validates each record against the ClickHouse schema for `{table}`, and publishes it to the message queue. Returns immediately — ClickHouse insertion happens asynchronously via the batch consumer. +Validates a body of records against the ClickHouse schema for `{table}` and publishes each accepted one to the message queue. Returns immediately — ClickHouse insertion happens asynchronously via the batch consumer. A single-object body answers `{"ok":true}` (or `{"duplicate":true}` when dedup is on); every other body answers the [batch summary](#batch-ingest). -**`Content-Type` is required and authoritative.** The format is what the caller declares, not what the bytes look like: a body declared as NDJSON is read as NDJSON whatever its first byte, so a line that isn't a JSON object fails as a per-record error rather than silently re-framing the whole request. A request with **no** `Content-Type`, or one whose media type is not in the accepted list, is rejected with `415` and a message listing the accepted types — nothing is guessed. The one thing the body still decides is *arity within the JSON family*: the first non-whitespace byte picks a top-level array (`[`) or a single object. The reverse mis-declaration is **not** caught: NDJSON sent as `application/json` is read as the single object it starts with and the remaining lines are ignored — a `200` for one record. Declare `application/x-ndjson` for anything line-framed ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). - -:::note[What counts as a valid declaration] +**The body goes to ClickHouse's own parser as-is.** WaveHouse never decodes a record: validation, type coercion, `DEFAULT` substitution and timestamp parsing are ClickHouse's own, running in-process via [chtypes](/deployment#chtypes-artifacts) (`internal/typelayer`) — the exact code path a real `INSERT` runs. A rejection therefore carries ClickHouse's own message and its numeric error number as `exception_code` (the string `code` stays the failure class, as on the query paths) rather than a WaveHouse-authored sentence, and there is no separate coercion table to keep in sync with the server. -The header is parsed with Go's `mime.ParseMediaType`, which implements RFC 9110 §8.3 `media-type`, and only the media type decides the format. Parameters are ignored, so no malformed parameter costs you the request — `application/json; charset`, `application/json;;`, a value left mid-quote, even a name repeated with different values all read as `application/json`. One exception: a malformed parameter on a line that **also contains a comma** is refused, because the comma may be a second declaration joined on and the error cannot distinguish that from a comma inside data ([#563](https://github.com/Wave-RF/WaveHouse/issues/563)). So `application/json; profile="a,b"` is one media type and is accepted, while `application/json; profile="a,b"; charset` is a `415` — each half alone is fine. +**`Content-Type` is required and authoritative**: it declares the format and the bytes never override it. -`Content-Type` is also a **singleton** field, and §5.3 forbids repeating it. So anything that isn't exactly one readable media type is a `415`: no header, an unsupported type, one whose **media type** doesn't parse, or more than one declaration. The single accommodation is for intermediaries that duplicate the header — **repeated header lines** are all resolved and accepted when they agree on the format. A **comma-joined** value is not — §8.3 warns that picking a member of the resulting pseudo-list is itself an interoperability and security hazard. Precisely: a value carrying a comma is refused whenever the value as a whole does not parse as one media type — whatever made it unparseable. A comma *inside a quoted parameter value* is legal data, so `application/json; a=", application/x-ndjson; b="` is one media type and is accepted, even though an intermediary may have built it by illegally joining two lines — the server cannot tell. +| `Content-Type` | Body | +| --- | --- | +| `application/json` | one flat object, **or** a top-level array of them | +| `application/x-ndjson`, `application/ndjson`, `application/jsonl`, `application/jsonlines` | one object per line; always a batch | +| `text/csv`, `text/tab-separated-values` | positional, with ClickHouse's own header auto-detection — see [Positional formats](#positional-formats-csv--tsv) | +| `text/csv; header=absent`, `text/tab-separated-values; header=absent` | strictly positional, no header detection | +| `text/csv; header=present`, `text/tab-separated-values; header=present` | a header line naming the columns, in any order — see [Header formats](#header-formats-headerpresent) | +| anything else, or none | `415`, listing the accepted types | -The 415 body quotes what you declared, bounded: at most **four distinct** header lines, each capped at 128 bytes and marked `…(truncated)` when cut, followed by `"…and N more"`. N counts every header line not quoted — *including duplicates of one that is* — so five copies of the same header show it once, then `"…and 4 more"`. When declarations conflict, the one that actually disagreed is always quoted, even when four agreeing spellings would otherwise fill the list. -::: +The two JSON families are one format to ClickHouse; the declaration decides only how the body frames its records. The single thing the body still chooses is *arity within `application/json`*: the first non-whitespace byte picks an array (`[`) or a single object. Under a single-object body only the first object is read — concatenated objects after it are ignored, a `200` for one record; declare NDJSON for anything line-framed ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). The reverse now works: a JSON array declared `application/x-ndjson` ingests every element. -| Body | `Content-Type` | Response | -| ---- | -------------- | -------- | -| one flat JSON object | `application/json` | `{"ok":true}` (or `{"duplicate":true}`) | -| a JSON array of objects (any length, even 1) | `application/json` | per-record summary — see [Batch Ingest](#batch-ingest) | -| one JSON object per line (NDJSON) | `application/x-ndjson` (also `application/ndjson`, `application/jsonl`, `application/jsonlines`) | per-record summary — see [Batch Ingest](#batch-ingest) | +:::note[What counts as a valid declaration] +The header is parsed with Go's `mime.ParseMediaType` (RFC 9110 §8.3) and the **media type** decides the format, so no malformed *parameter* costs the request — `application/json; charset`, `application/json;;`, a value left mid-quote, a name repeated with different values all read as `application/json`. The one parameter that also decides a format is `header`, on `text/csv` and `text/tab-separated-values` only: `present` selects the header format, `absent` the strictly positional one, no `header` at all ClickHouse's default reading, and any other value is a `415`. A line whose parameters did not parse and that mentions `header` is a `415` as well, because guessing at it could ingest a declared header line as data or drop a data row as a header. Two more things are refused. A malformed parameter on a line that **also contains a comma** is a `415`, because the comma may be a second declaration joined on and the error cannot tell that from a comma inside data ([#563](https://github.com/Wave-RF/WaveHouse/issues/563)) — so `application/json; profile="a,b"` is fine and `application/json; profile="a,b"; charset` is not. And `Content-Type` is a **singleton** field (§5.3 forbids repeating it), so repeated header *lines* are accepted only when they agree, while a comma-joined value is refused outright: §8.3 warns that picking a member of the resulting pseudo-list is itself an interoperability and security hazard. -The inbound request body is capped at 16 MiB; a body over the cap is rejected with `413` (matching [`POST /v1/ops/query`](#post-v1opsquery--query-clickhouse)). The cap applies to **every** body shape, NDJSON included — the whole body is read before it is parsed, so a line-framed batch is bounded by the same 16 MiB cap as a JSON array (NDJSON carries one additional, tighter bound: a single line over 10 MiB fails the request). The `413` is decided before any record is processed, so nothing is published — including for a single-object body whose trailing bytes push it over the cap, which is now rejected rather than accepted on its first object. Split an upload larger than the cap across several requests, and set your own outer limit at the [reverse proxy](/reverse-proxy#request-body-size-limits). +The 415 body quotes what you declared, bounded: at most **four distinct** header lines, each capped at 128 bytes and marked `…(truncated)` when cut, then `"…and N more"` counting every line not quoted, duplicates included. When declarations conflict, the one that actually disagreed is always quoted. +::: -The `{table}` URL query must match a table that exists in ClickHouse. WaveHouse discovers table schemas on startup and refreshes them periodically. +The inbound request body is capped at 16 MiB and the `413` is decided before any record is processed, so nothing is published. The cap applies to every body shape — the whole body is read before it is parsed, so a line-framed batch is bounded exactly as a JSON array is. Split a larger upload across several requests, and set your own outer limit at the [reverse proxy](/reverse-proxy#request-body-size-limits). The `{table}` query parameter must name a table WaveHouse has discovered in ClickHouse; schemas refresh periodically. :::note[Insert-only] The ingest pipeline accepts only inserts. All other mutations — `DELETE`, `UPDATE`, `TRUNCATE`, `DROP`, `ALTER`, `REPLACE`, etc. — must be issued through [`POST /v1/ops/query`](#post-v1opsquery--query-clickhouse), which is restricted to the admin role (`admin_role`, the same gate as the rest of `/v1/ops/*`), or through an operator-authored [pipe that writes](/pipes#pipes-that-write): an operator authors its statement in `pipes.json`, and the roles in its `allowed_roles` run it with parameter values only. @@ -256,60 +251,32 @@ The ingest pipeline accepts only inserts. All other mutations — `DELETE`, `UPD The policy engine authorizes mutations by inspecting the columns being written. That works for inserts but not for predicate-driven mutations like `DELETE … WHERE` — there's no way to prove the predicate matches only rows the caller is allowed to touch. Routing those statements through the admin-gated raw-SQL surface, or through a pipe whose predicate the operator wrote, keeps the policy contract honest. ::: -**Request:** +**What ClickHouse decides, and what WaveHouse decides.** Everything about a *value* is ClickHouse's: -```json -{ - "url": "https://example.com/dashboard", - "user_name": "Alice", - "verified": true, - "score": 42.5 -} -``` +- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED`, `ALIAS` and `EPHEMERAL` columns — none of the three is ever part of a published row. +- An omitted column, or an explicit `null` on one (WaveHouse pins `input_format_null_as_default`), takes its `DEFAULT` expression — evaluated by ClickHouse, including a volatile one like `now()` — or the type's implicit zero where none is declared, exactly as an `INSERT` naming fewer columns does. +- A coercion ClickHouse would make it makes here (a numeric string into an `Int*`, `"true"` into a `Bool`, an out-of-range integer wrapping); anything it would refuse fails synchronously in the ingest response with its real code, rather than surfacing later in the DLQ. `Nullable()` and `LowCardinality()` wrappers are transparent. -The body is a **flat JSON object** whose keys must match column names in the target ClickHouse table. Values must be type-compatible (see schema validation below). +WaveHouse decides only policy: whether the role may insert at all, and whether the record satisfies the role's [`check` clauses](/access-control#insert-checks) — evaluated by the same compiled-filter engine as row-level security, in the same parse that validates the record and against the row ClickHouse produced, so a check sees stored values rather than the payload's spelling. A record ClickHouse refuses reports that refusal, never a check result. A record chtypes cannot evaluate at all — as opposed to accepting or rejecting it — is **declined** (`422`), which is not a data verdict. -**Schema Validation:** - -- Unknown fields (not in the ClickHouse schema) are rejected. -- Type mismatches are rejected (e.g., sending a boolean for a `Float64` column). -- Missing required columns (non-nullable without a default) are rejected. -- Null values for non-nullable columns without a default are rejected. -- A value for a `MATERIALIZED` or `ALIAS` column is rejected — ClickHouse computes those, and the published row has no slot for one. -- **An omitted `Nullable(T) DEFAULT …` column now stores `NULL`, not the default.** The published row is positional, with one slot per insertable column and no way to express "absent", so an omitted key rides as an explicit `null`; `input_format_null_as_default` rescues that only for a **non-nullable** column. On a nullable column only an absent key ever took the default, and a positional row cannot express absence. Verified on ClickHouse 26.6.3. (One case changes only on a server explicitly running `input_format_null_as_default=0`: an explicit `null` for a non-nullable column with a default now takes the default there rather than failing the row into the DLQ, because WaveHouse pins the setting instead of inheriting it. On a default-configured server this was already the behavior.) -- Type compatibility: `String` accepts JSON strings, numbers, and booleans (ClickHouse coerces the non-strings); `FixedString`/`UUID` accept the same at validation, but ClickHouse rejects a non-string value there, so it surfaces in the DLQ; `DateTime`/`Date`/`Enum` accept JSON strings or numbers; `IPv*` accepts JSON strings (a number passes validation but ClickHouse rejects it → DLQ); `Int*`/`Float*`/`Decimal` accept JSON numbers or strings — a string lets JavaScript callers avoid 64-bit precision loss, and its contents are ClickHouse's to judge (a non-numeric string is accepted here and surfaces in the DLQ, not as a `400`); `Bool` accepts JSON booleans and the numbers `0`/`1` (any other number, and *any* string — including `"true"` — passes validation but is rejected by ClickHouse → DLQ); `Array` accepts JSON arrays; `Map` accepts JSON objects; `Tuple` accepts JSON arrays or objects at validation, but ClickHouse takes an array only for an *unnamed* tuple and an object only for a *named* one — the other shape surfaces in the DLQ; any other ClickHouse type (`JSON`, `Variant`, `Dynamic`, geo, …) accepts any JSON value — WaveHouse defers to ClickHouse, so a bad value surfaces in the DLQ rather than as a `400`. -- `Nullable()` and `LowCardinality()` wrappers are handled transparently. -- Top-level `DateTime`/`DateTime64` values are rewritten to a canonical wire form on ingest — see [Timestamp canonicalization](#timestamp-canonicalization). - -**Response (accepted):** - -```json -{"ok": true} -``` - -**Response (duplicate):** *(only when dedup is enabled)* - -```json -{"duplicate": true} -``` - -**Error responses:** +**Error responses.** Rows marked **per-record** are reported in `results` on a batch body (the request itself stays `200`) and become the response status on a single-object body; every other row fails the whole request. | Status | Body | Cause | | ------ | ---- | ----- | +| 400 | `{"error":"","exception_code":}` | **Per-record.** ClickHouse's parser refused the record; `exception_code` and the message are its own. `117` is an unknown field — which now includes a column the role may not write, and any `MATERIALIZED`/`ALIAS`/`EPHEMERAL` column; `27`/`26` are unparseable input; `6` out of range | +| 400 | `{"error":"Unknown field found in format header: 'x' at position 1 …","code":"clickhouse.rejected","exception_code":117}` | A `header=present` body whose header names a column the table — or the role's writable set — does not have, or names one twice. ClickHouse refuses the body before reading any record, so the whole request fails and nothing is published | | 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is `invalid json` | -| 400 | `{"error":"invalid json"}` | Malformed request body | -| 400 | `{"error":"unknown column ... for table ..."}` (also: `missing required column ...`, `type mismatch for column ...`, `null value for non-nullable column ...`) | Schema validation failure (unknown fields, type mismatches, missing required columns, null in a non-nullable column with no default). The body is the validator's message verbatim — there is no `validation failed:` prefix. | -| 400 | `{"error":"column \"x\" of table \"t\" is materialized and cannot be inserted"}` (also `… is alias …`) | The record supplies a value for a column ClickHouse computes. Omit it — the server fills it in. Refused rather than dropped: the published row has one slot per insertable column, so the value would otherwise vanish behind a `200` | -| 400 | `{"error":"missing dedupe id field \"event_id\""}` | Only when dedupe is enabled with `dedupe.require_id: true` and the row lacks the configured `id_field` or sets it to `null`. With `require_id: false` (the default) the row is instead published un-deduped. Either way — reject or publish — the row is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`. In a batch this is a per-record failure, not a whole-request error. | +| 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | +| 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | +| 400 | `{"error":"missing dedupe id field \"event_id\""}` | **Per-record.** Only with `dedupe.require_id: true`, when the row carries no value for the configured `id_field` (an absent column or a `null` cell). With `require_id: false` (the default) the row is published un-deduped instead. Either way it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (the gate surfaces the token reason rather than silently falling back to `default_role`) | -| 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table | -| 403 | `{"error":"column \"x\" not allowed for insert"}` | The record names a column the role's `allow_columns`/`deny_columns` forbids ([Access control → Column permissions](/access-control#column-permissions)) | -| 403 | `{"error":"check failed for column \"x\""}` | The record's value for a checked column doesn't satisfy the policy `check` (`_eq`/`_in`), or an `_in`-checked column is omitted ([Access control → Insert checks](/access-control#insert-checks)). On the batch path both this and the column error above are per-record failures reported in `results`, not whole-request rejections | -| 403 | `{"error":"policy check references column \"x\", which table \"t\" does not have"}` (also `… which is materialized and cannot be inserted`, the same for `alias`, and `… which is ephemeral and is never stored`) | A **policy misconfiguration**, not a bad request: the role's `check` names a column the table lacks, one ClickHouse computes, or an `EPHEMERAL` one. None can be enforced — the published row carries one slot per insertable column, and an ephemeral column is never stored — so the check would have passed silently while enforcing nothing. Like the rejections above this is decided per record, so the **status depends on the body shape**: a single-object request answers `403`, while a batch answers `200` and carries the same message against each record in `results`. It fires on **every** insert by that role until the policy or the table is corrected, and names every offending column rather than one of them. `wavehouse validate` cannot catch it: it never sees the ClickHouse schema | +| 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table — checked once, before any record | +| 403 | `{"error":"check failed for column \"x\""}` (several: `check failed for columns "x", "y"`) | **Per-record.** The row does not satisfy the role's insert [`check`](/access-control#insert-checks). The filter is AND-joined over every checked column, so with more than one it names the set that was tested rather than guessing an attribution | +| 403 | `{"error":"policy check references column \"x\", which table \"t\" does not have"}` (also `… which is materialized and cannot be inserted`, the same for `alias`, and `… which is ephemeral and is never stored`) | **Per-record.** A **policy misconfiguration**, not a bad request: the role's `check` names a column the table lacks, one ClickHouse computes, or an `EPHEMERAL` one. None can be enforced, so the check would have passed silently while enforcing nothing. It fires on every insert by that role until the policy or the table is corrected, and names every offending column. `wavehouse validate` cannot catch it — it never sees the ClickHouse schema | | 404 | `{"error":"unknown table: ..."}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | -| 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, …"}` (declared variant: `Content-Type "text/plain": ingest requires one of …` — see the note above on how declarations are echoed; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | The request declared no `Content-Type`, one whose media type is unsupported or does not parse, a comma-bearing value that does not parse as a single media type, or repeated header lines that disagree — different formats, or one supported and one not. Checked before the body is parsed | +| 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | +| 422 | `{"error":"validation engine declined: "}` | **Per-record.** chtypes could not evaluate the record at all — the artifact declined the shape, rather than the data being wrong. A `check` clause that could not be evaluated lands here too (`validation engine declined: the insert check for column "x" could not be evaluated`) | | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now: it is not open (for example, it failed to open on a reload), or a DynamoDB table is throttling, timing out or unreachable; `Retry-After: 5`. Nothing was published, so the retry is safe | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet (its ClickHouse unreachable, or [no pool for it](/settings-directory#clickhouse)), so whether the table exists is not known; `Retry-After: 5`. Decided before the body is read | @@ -317,6 +284,7 @@ The body is a **flat JSON object** whose keys must match column names in the tar | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure, for that tenant alone) or not open (see [Message Queue](/settings-directory#message-queue)). Under [`mq.backend: nats`](/deployment#external-nats), the table's partition stream is full, which refuses every table in it, or the tenant's table holds as many unwritten rows as the stream allows one subject. Response includes `Retry-After: 30` header. With dedupe on, the record's id is given back, so the retry is published rather than reported as a duplicate. | | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`, a transient broker failure, not a full queue). Only under [`mq.backend: nats`](/deployment#external-nats), including a partition stream the operator deleted; the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the record's id is left to lapse rather than given back, so a retry cannot land as a second copy; `Retry-After` is that lease, rounded up to whole seconds, when dedupe was on for the record, else the flat `Retry-After: 5`. | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | Dedupe is on and another request carrying the same id is still being published — usually a client's timeout-retry racing its own original. Its outcome decides whether this record is a duplicate, so retry after the `Retry-After` header (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). | +| 503 | `{"error":""}` | The tenant's ClickHouse line has no installed chtypes artifact for its server version, or its server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line). The body names the cause. Only that tenant is refused — every other tenant keeps working — and it is decided before the body is read, with `Retry-After: 5`. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | **curl example:** @@ -325,73 +293,82 @@ The body is a **flat JSON object** whose keys must match column names in the tar curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ -H "Content-Type: application/json" \ -d '{"page": "/home", "button": "signup", "score": 42.5}' +# → {"ok":true} ``` -#### Timestamp canonicalization +#### Timestamp rendering -**Send the canonical form — RFC 3339 UTC with the fraction already truncated to the column's precision and trailing zeros trimmed (spelled out below) — and the value is republished byte-for-byte.** A `DateTime`/`DateTime64` column value sent in any other accepted form — including RFC 3339 UTC with extra or trailing-zero fraction digits — is rewritten to that **one canonical wire form — RFC 3339 UTC** (`2026-06-21T04:00:00Z`, fraction truncated to the column's precision) before publishing, so the stored instant never changes but every consumer (the ClickHouse insert, [SSE subscribers](#get-v1stream--server-sent-events-stream), the DLQ) sees the same spelling `/v1/query` renders. The accepted input forms: +WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling ClickHouse's own parser accepts under `date_time_input_format=best_effort` (the setting WaveHouse pins, both at chtypes' ingest compile and on the worker's `INSERT`) is accepted — RFC 3339 with any offset, a zone-less `YYYY-MM-DD[ T]HH:MM:SS[.fff]` read in the column's declared zone else the server's default, a Unix-seconds string, a bare integer at the column's tick scale, among the other forms its lenient parser reads. It is ClickHouse's grammar, not a reimplementation of it, so whatever a real `INSERT` into this table would accept, ingest accepts, with the same coercions and the same refusals. -- RFC 3339, any offset (`.`-fractions only — ClickHouse has no `,` separator). -- `YYYY-MM-DD[ T]HH:MM:SS[.fff]` or `YYYY-MM-DD`, zone-less — interpreted in the column's time zone, else the ClickHouse server's, exactly as ClickHouse itself would. -- A Unix-seconds string of exactly 9–10 digits (a `.fff` fraction is honored only for `DateTime64` columns, as ClickHouse does). -- A **non-negative integer** JSON number, unquoted — read the way ClickHouse reads bare numbers: Unix **seconds** for a `DateTime` column, but the column's raw **tick count** for `DateTime64` (a `DateTime64(3)` stores milliseconds, so `1750478400500` is the millisecond epoch `2025-06-21T04:00:00.5Z` — and `1750478400` is January 1970, not June 2025). +**Outbound**, `DateTime`/`DateTime64` values in the NATS/SSE wire row and in `/v1/query` / `/v1/pipes/{name}` results are the exact bytes ClickHouse's writer produces, in the column's declared zone else the server's default: `"2026-06-21 04:00:00.123"`, space-separated, no `Z` suffix, never RFC 3339. Every consumer renders from the same stored value the same way, so SSE and `/v1/query` agree on spelling for a given row **by construction**, with no WaveHouse rewriting step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). The raw-SQL proxy `/v1/ops/query` is the exception: it sets `date_time_output_format=iso`, which keeps trailing fraction zeros and an ISO-8601 `Z`, and is not expected to match the other two byte-for-byte. -**Fail-open**: a value in none of those forms is published verbatim — ClickHouse's more liberal parser decides insertability, and a value it too rejects surfaces via the DLQ, as before. `Date`/`Date32` columns pass through untouched. +**One server time zone per ClickHouse line, per process.** The in-process parser takes its zone once, when a process first opens a ClickHouse line, and keeps it for the process's lifetime. A tenant whose server reports a different zone than the one that line was opened with is refused on its own — ingest answers `503`, its stream rows are withheld with reason `unavailable` — while every other tenant keeps working. Run tenants whose servers use different zones in separate processes. -:::note[Pass-through edge cases] +**Row-level security compares instants, not spellings.** A stream row filter on a `DateTime`/`DateTime64` column is compiled and evaluated by the same engine that validates ingest ([internal/typelayer](/access-control#where-each-rule-is-enforced)), so a filter constant in any spelling ClickHouse would accept in a `WHERE` clause matches the stored instant however the payload spelled it, and a predicate the engine cannot compile withholds every row for that role. -- Digit-strings of lengths other than 9–10 are ClickHouse's own forms — calendar shapes like `YYYYMMDD`, or its 13/16/19-digit ms/µs/ns epochs — and pass through untouched. -- A bare number with a fraction or exponent (`1750478400.5`) is passed through un-rewritten, and ClickHouse then fails the row for any timestamp column: it parses bare numbers as integers only — its lenient timestamp parsing, which accepts `"1750478400.5"` for a `DateTime64` column (on a plain `DateTime` the leftover fraction still fails the row), applies solely to quoted strings. -- An instant outside the column type's range also passes through — ClickHouse *saturates* out-of-range values spelling-dependently (and a `DateTime64(9)` column rejects the insert outright past the Int64-nanosecond ceiling, 2262-04-11 — a bound WaveHouse conservatively applies to every `DateTime64` of precision ≥ 7 when deciding what it may rewrite), so no rewrite there is safe. -- A time zone that doesn't resolve at runtime also causes pass-through: the binary embeds no tzdata, so named zones resolve from the runtime's zone database (the bundled distroless images ship one; a stripped-down custom runtime, or a server zone newer than the image's snapshot, may not resolve). An unresolvable *column* zone skips canonicalization for that column entirely; an unresolvable *server* default skips only zone-less values of columns without a declared zone — warned at schema refresh either way, and never guessed as UTC, which could move the stored instant. Remedy: install `tzdata` in a custom image, or point Go at a zone database via the `ZONEINFO` environment variable. -- Timestamps nested inside a composite column (`Array(DateTime)`, `Map(K, DateTime64)`, `Tuple(…, DateTime)`) pass through untouched; only top-level `DateTime`/`DateTime64` columns (including `Nullable`/`LowCardinality` wrappers) are canonicalized. -- The accepted grammar is differentially tested against a live ClickHouse: raw and canonicalized spellings must insert identically, or both fail. +#### Positional formats (CSV / TSV) -::: +`text/csv` and `text/tab-separated-values` are **positional**; `header` is [RFC 4180 §3](https://www.rfc-editor.org/rfc/rfc4180#section-3)'s optional parameter, and WaveHouse maps it onto ClickHouse's own behavior three ways: -:::caution[Upgrading WaveHouse against a pre-26.5 ClickHouse] -WaveHouse pins `date_time_input_format=best_effort` on its inserts — the ClickHouse server default since 26.5. On an older server whose default was `basic`, a plain `DateTime` column read an all-digit timestamp string of five or more digits as Unix seconds (shorter runs it rejected outright, where `best_effort` reads `"2026"` as a year); under `best_effort`, `"20260711"` stores 2026-07-11, not 1970-08-23, and some lengths (e.g. 12 digits) are rejected outright. (`DateTime64` columns diverge the same way on calendar-shaped runs — `"20260711"` is 1970-08-23 under `basic`, 2026-07-11 under `best_effort` — and additionally whenever an epoch run's unit doesn't match the column scale, e.g. a 16-digit microsecond epoch into a `DateTime64(3)`; an epoch run whose unit matches the column scale (a 13-digit millisecond epoch into a `DateTime64(3)`) reads identically too — only 9–10-digit Unix-seconds runs, with an optional fraction, agree at *every* scale.) The canonical form itself is what the pin rescues: under `basic` an RFC 3339 value's `Z` suffix is rejected outright (the row fails and lands in the DLQ), and the pin is what makes it insertable regardless of server version. Zone-less date-times and 9–10-digit Unix-seconds strings parse identically under both settings. -::: +| Content-Type | Reading | +| --- | --- | +| `text/csv; header=present` | `CSVWithNames`: a header line is required and columns are addressed by name — see [Header formats](#header-formats-headerpresent) | +| `text/csv; header=absent` | `CSV`, strictly positional: header detection is off (`input_format_csv_detect_header=0`), so every line is a record | +| `text/csv` (no `header` parameter) | ClickHouse's default `CSV`: header auto-detection stays on | -**The canonical form, precisely.** This is the one strict timestamp spelling in WaveHouse — the same one `/v1/query` and `/v1/pipes/{name}` render for top-level timestamp columns and the SSE stream carries (the raw-SQL proxy `/v1/ops/query` instead renders server-side via `date_time_output_format=iso`, which keeps trailing fraction zeros): +`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. With no parameter, the positional fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three computed kinds and that is the field order. -- `YYYY-MM-DDTHH:MM:SSZ`, or `YYYY-MM-DDTHH:MM:SS.FZ` when there is a sub-second part: uppercase `T` separator, uppercase `Z` suffix, always UTC — never a numeric offset — and seconds always present. -- The fraction is **truncated** (never rounded) to the column's precision: a `DateTime` column (whole seconds) never carries a fraction; a `DateTime64(3)` column carries at most three digits. -- Trailing fractional zeros are trimmed and an all-zero fraction is dropped (Go's `time.RFC3339Nano` rendering): `.120` becomes `.12Z`, `.000` becomes plain `Z` — byte-for-byte what `/v1/query` returns for the same stored value. -- A column's declared time zone changes only how zone-less *inputs* are interpreted, never the output: every canonical value ends in `Z`. +| Body | Outcome | +| --- | --- | +| every field, in order | accepted | +| an empty field (CSV) or `\N` (TSV) | that column takes its `DEFAULT` | +| too few fields | rejected, code **27** — ClickHouse's own message, e.g. `Cannot parse input: expected ',' before: …` | +| too many fields | rejected, code **117** — `Expected end of line` | +| a header line, no `header` parameter | ClickHouse detects it and **consumes** it as a header: `total` and every `index` count data rows only | +| a header line, `header=absent` | **not a header** — read as a data row, so it fails to parse (code 27) wherever a column cannot read its own name; the data rows after it still parse | -Examples for a `DateTime64(3, 'America/New_York')` column: `"2026-06-21 00:00:00.1239"` (zone-less, read in New York) → `"2026-06-21T04:00:00.123Z"`; `"1750478400.5"` (Unix-seconds string) → `"2025-06-21T04:00:00.5Z"`; the integer number `1750478400500` (ticks at the column's millisecond scale) → `"2025-06-21T04:00:00.5Z"`. +The messages are ClickHouse's own and differ between ClickHouse lines; branch on the `exception_code`. An empty **TSV** field is the empty string, not a default: `\N` is TSV's spelling for "take the default", and a `DateTime64` cannot read `""`. -**The stream row-filter doesn't require this spelling.** Row-level enforcement compares timestamp operands as **instants** under the same input grammar, so a filter constant in any accepted spelling — zone-less, RFC 3339, Unix seconds — matches the canonical payload denoting the same instant, and an operand the grammar can't read withholds the row. Instant comparison also needs the column's timestamp parser from schema discovery — with no usable schema, or a declared zone that can't be loaded at runtime, the column falls back to byte-equality, where only an exactly matching spelling admits. See [the enforcement caution](/access-control#where-each-rule-is-enforced) for per-type comparison rules and the spelling that also works in query-path SQL. +A positional producer cannot self-describe, so a column-order change silently re-assigns values — pin it to the schema and re-check it after any `ALTER`, or send a header with `header=present`. -#### Batch Ingest - -A **JSON array** of objects (`[{…}, {…}]`) or an **NDJSON** body (`Content-Type: application/x-ndjson`, one JSON object per line) ingests a batch in a single request. Each record is validated, authorized, deduplicated, and published independently, so **one malformed or rejected record never blocks the rest of the batch**. (The SDK's `insert([...])` array helper uses the NDJSON form automatically; both forms return the same response.) +```bash +curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ + -H "Content-Type: text/csv" \ + --data-binary $'"/home","signup",42.5,\n"/about","nav",3,\n' +# → {"total":2,"succeeded":2,"failed":0,"duplicates":0,"results":[{"index":1,"ok":true},{"index":2,"ok":true}]} +``` -- **JSON array** — the most convenient form from most HTTP clients. A structural JSON syntax error fails the whole request (`400`), but a wrong-typed element (a non-object) is reported per-record like any other rejection. An explicit empty array (`[]`) is a valid, record-less batch (`200`, `total: 0`). -- **NDJSON** — one record per line. Its advantages are tolerance of a malformed record and cheap client-side generation, not a larger ceiling: the body cap applies to it exactly as to a JSON array. Blank lines are skipped, and a single malformed *line* is reported and skipped (the newline reframes the next record) — where a structural syntax error anywhere in a JSON array fails the whole request. Both forms report a wrong-typed record per-record. +#### Header formats (`header=present`) -**Request (JSON array):** +`text/csv; header=present` and `text/tab-separated-values; header=present` open with a header line naming the columns, and the fields are addressed by it rather than by position. IANA's `text/tab-separated-values` registration defines no parameters, so `header` on TSV is WaveHouse's mirror of the CSV one. -```http -POST /v1/ingest?table=clicks -Content-Type: application/json +| Body | Outcome | +| --- | --- | +| a header naming the columns, in any order | the header is **not a record**: `total` and every `index` count data lines only | +| a column the header omits | takes its `DEFAULT`, exactly as an omitted JSON field does — including a check clause's injected value | +| a header name in a different case | matched case-insensitively, as ClickHouse does from 26.5 | +| a header naming a column the table, or the role's writable set, lacks — or a name given twice | the whole request is a `400` with code **117**; nothing is published | +| a bad data row | rejected per record with its code; the rows around it still ingest | +| only the header line | a valid record-less batch: `200`, `total: 0` | -[{"page": "/home", "button": "signup", "score": 42.5}, {"page": "/about", "button": "nav", "score": 3}, {"page": "/pricing", "button": "cta", "score": 7, "referrer": "/home"}] +```bash +curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ + -H "Content-Type: text/csv; header=present" \ + --data-binary $'button,page\nsignup,/home\nnav,/about\n' +# → {"total":2,"succeeded":2,"failed":0,"duplicates":0,"results":[{"index":1,"ok":true},{"index":2,"ok":true}]} ``` -**Request (NDJSON):** +#### Batch Ingest -```http -POST /v1/ingest?table=clicks -Content-Type: application/x-ndjson +A JSON array, an NDJSON body, a CSV body or a TSV body ingests a batch in one request. Each record is validated, authorized, deduplicated and published independently, so **one malformed or rejected record never blocks the rest of the batch** — including inside a single-line (compact) JSON array, which WaveHouse re-frames in place before handing it over. An explicit empty array (`[]`) is a valid record-less batch (`200`, `total: 0`); blank lines in an NDJSON body are skipped. (The SDK's `insert([...])` array helper uses the NDJSON form automatically; every form returns the same response.) -{"page": "/home", "button": "signup", "score": 42.5} -{"page": "/about", "button": "nav", "score": 3} -{"page": "/pricing", "button": "cta", "score": 7, "referrer": "/home"} -``` +The response counts records read, published, rejected and deduplicated, then lists per-record outcomes: each entry mirrors the single-object response (`ok` / `duplicate` / `error`) plus its 1-based `index`, and carries ClickHouse's numeric `exception_code` when the rejection was its parser's. `results` is truncated to the first 10,000 entries; the four counts stay authoritative. -**Response (`200`):** a per-record summary. Each `results` entry mirrors the single-object response (`ok` / `duplicate` / `error`) plus its 1-based `index`. +```bash +curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ + -H "Content-Type: application/json" \ + -d '[{"page":"/home","button":"signup","score":42.5},{"page":"/about","button":"nav","score":3},{"page":"/pricing","button":"cta","score":7,"referrer":"/home"}]' +``` ```json { @@ -402,7 +379,7 @@ Content-Type: application/x-ndjson "results": [ { "index": 1, "ok": true }, { "index": 2, "ok": true }, - { "index": 3, "error": "unknown column \"referrer\" for table \"clicks\"" } + { "index": 3, "error": "Unknown field found while parsing JSONEachRow format: referrer", "exception_code": 117 } ] } ``` @@ -415,44 +392,29 @@ Content-Type: application/x-ndjson | `duplicates` | records skipped by dedup (when enabled) | | `results` | per-record outcomes, each `{ index, ok\|duplicate\|error }` with `index` the 1-based record position. Truncated to the first 10,000 entries for very large batches (the counts stay authoritative). | -A `200` is returned whenever the body was read and the records were processed — **even if every record failed**, so branch on `failed`/`results`, not the status code. Per-record problems (a malformed NDJSON line, a non-object array element, schema validation, column/check permission failures) are reported in `results` and the batch continues. Whole-request conditions abort with a non-`200` instead: +A `200` is returned whenever the body was read and the records were processed — **even if every record failed**, so branch on `failed`/`results`, not the status code. Per-record problems (a malformed NDJSON line, a non-object array element, a ClickHouse parser rejection, a denied column or a failed `check`) are reported in `results` and the batch continues. Whole-request conditions abort with a non-`200` instead: | Status | Body | Cause | | ------ | ---- | ----- | -| 400 | `{"error":"empty body"}` / `{"error":"empty ndjson body"}` | The body has no records | +| 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | | 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is `invalid json` | -| 400 | `{"error":"invalid json: ..."}` | A structural JSON syntax error, a single NDJSON line over 10 MiB, or a JSON array that ends before its closing `]` — a body that transferred completely but was generated truncated. The whole request fails rather than reporting a partial success | +| 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (same auth gate as the single-object path; surfaces the token reason) | | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table (checked once, before any record) | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | -| 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, …"}` (declared variant: `Content-Type "text/plain": ingest requires one of …` — see the note above on how declarations are echoed; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | The request declared no `Content-Type`, one whose media type is unsupported or does not parse, a comma-bearing value that does not parse as a single media type, or repeated header lines that disagree — different formats, or one supported and one not. Checked before the body is parsed | +| 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | | 500 | `{"error":"publish failed"}` / `{"error":"dedupe failed"}` | Message-queue or dedup-backend failure mid-batch, other than a full queue or an unreachable broker (below). After a publish failure the records before it keep their ids, so a whole-batch retry reports those as duplicates; the failing record's id is left to lapse as on the single-object path, and the rest of its window's ids are given back | | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure) or not open, mid-batch; includes `Retry-After: 30`. The records before the refused one keep their ids, and its id and the rest of its window's are given back | | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`), mid-batch. Only under [`mq.backend: nats`](/deployment#external-nats); the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the failing record's id is left to lapse rather than given back — so `Retry-After` is that record's dedupe lease, rounded up to whole seconds, when it was deduped; a record published un-deduped has no lapsing claim to wait out, so `Retry-After: 5` | | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now; `Retry-After: 5`. Nothing in the window being reserved was published; the windows before it were, and keep their ids | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | A record's dedupe id is held by another request still being published; includes `Retry-After` (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). Nothing in that record's window was published; the windows before it were | +| 503 | `{"error":""}` | The tenant's ClickHouse line has no installed chtypes artifact, or its server time zone differs from the zone this process opened that line with — decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] -A batch aborted partway (a `503`/`500`, a JSON-array syntax error, or an NDJSON line over the 10 MiB line bound, after some leading records were already published) re-publishes those leading records when the whole batch is retried. Records are published in windows of 256, in order: a read error or a dedupe failure drops the open window unpublished, so what an aborted batch published is the windows before it, plus, after a publish failure, the records of its window before the failing one. A whole-body read failure is **not** one of these: a `413`, or the `400 invalid request body` of an upload cut off in transit, is decided before any record is processed, so nothing is published — safe to retry, once split for a `413`. Enable deduplication if duplicate suppression matters — this is the same at-least-once property the single-object path already has (the SDK retries both on `503`). +A batch aborted partway — a `503` or `500` after some leading records were already published — re-publishes those leading records when the whole batch is retried. Records are published in windows of 256, in order: a dedupe failure drops the open window unpublished, so what an aborted batch published is the windows before it, plus, after a publish failure, the records of its window before the failing one. Failures decided before any record is processed are not in this class: a `413`, a `415`, the `400 invalid request body` of an upload cut off in transit, and the unterminated-array `400` all publish nothing and are safe to retry as-is (once split, for a `413`). Enable deduplication if duplicate suppression matters — the single-object path has the same at-least-once property, and the SDK retries both on `503`. ::: -**curl example (JSON array):** - -```bash -curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ - -H "Content-Type: application/json" \ - -d '[{"page":"/home","button":"signup","score":42.5},{"page":"/about","button":"nav","score":3}]' -``` - -**curl example (NDJSON):** - -```bash -curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ - -H "Content-Type: application/x-ndjson" \ - --data-binary $'{"page":"/home","button":"signup","score":42.5}\n{"page":"/about","button":"nav","score":3}\n' -``` - --- ### `POST /v1/ops/query` — Query ClickHouse @@ -571,19 +533,38 @@ Every column the query references — in `columns`, an aggregation argument, `fi | `columns` | string \| string[] | No | Columns to SELECT — an array, or a single string for one column. A literal `"*"` is the column *named* `*`, **not** a wildcard. Omit (or send `[]` / `""`) to select nothing; use `select_all` for a full-row read. Mutually exclusive with `select_all`. | | `select_all` | bool | No | Select every column the role may read (the all-columns wildcard, expanded server-side to the allow/deny set). Mutually exclusive with a non-empty `columns`, and with `aggregations`. | | `aggregations` | object[] | No | Aggregation functions (`fn`, `column`, `alias`). | -| `filters` | object[] | No | WHERE conditions (`column`, `op`, `value`). Ops: eq, neq, gt, gte, lt, lte, in, like. | +| `filters` | object[] | No | WHERE conditions (`column`, `op`, `value`). Ops: eq, neq, gt, gte, lt, lte, in, like. A `null` value is a `400`. | | `group_by` | string[] | No | GROUP BY columns. | | `order_by` | object[] | No | ORDER BY clauses (`column`, `dir`). | | `limit` | int | No | Max rows. Omitted or above the configured `query.default_max_rows` (default 10,000) → silently capped at that value; a policy `max_rows` can lower it further (see [Access Control](/access-control#resource-limits)). | | `time_range` | object | No | Time window (`column`, `since`, `until`). `since`/`until` accept RFC3339 or Go-duration relative values ("1h", "30m", "7d", "2w" — day and week suffixes expand to hours). Relative values mean that long *ago*. The window applies only when `column` and `since` are set — an `until` without `since` is ignored. | +:::note[Filter values bind as strings] +Every bound value — a caller's filter, a policy row filter, an insert `check` — binds as a `{p:String}` parameter and is compared under the column's own type. One rule across all three surfaces, and the answer is the server's: send ClickHouse's own spelling for a value and it reads it. A policy claim on an integer column is compared through a strict cast on top, so a claim that is not the canonical spelling of a value the column can hold matches nothing instead of wrapping (see [Access Control](/access-control#jwt-claim-templating)); a caller's own filter keeps the plain form, since it can only narrow what the policy admits. There is no RFC 3339 sniff any more: a value for a `DateTime` column is handed to ClickHouse verbatim, so a `"2026-01-01T00:00:00Z"` filter works exactly as far as the connected server's version reads that spelling. + +On **this endpoint** an `in` list binds as one `Array(String)` parameter, and ClickHouse caps a query-string parameter at about 64 KiB of literal text — roughly 6,000 short elements. Past that the query fails with ClickHouse's own `500` rather than a clean `400`; the 1 MiB request-body cap alone would have allowed more. Stream row filters bind their `in` elements one parameter each and carry no such ceiling. +::: + :::note[Identifier names] -Table, column, and alias names may contain any characters ClickHouse accepts — dots, spaces, unicode, reserved keywords — because every identifier is backtick-quoted automatically. The one exception is a name containing a literal `?`, which is rejected with `400` (a clickhouse-go positional-binder limitation tracked in [#279](https://github.com/Wave-RF/WaveHouse/issues/279)). +Table, column, and alias names may contain any characters ClickHouse accepts — dots, spaces, unicode, reserved keywords — because every identifier is backtick-quoted automatically. The one exception is a name containing a literal `?`, which is rejected with `400`: the builder assembles positional placeholders before rewriting them to ClickHouse's named parameters, and a `?` inside an identifier would desync that rewrite ([#279](https://github.com/Wave-RF/WaveHouse/issues/279)). ::: **Response:** -JSON array of result rows. Top-level `DateTime`/`DateTime64` values are returned in canonical RFC 3339 UTC (`2026-06-21T04:00:00.123Z`) — `Nullable` timestamp columns included (a SQL `NULL` renders as JSON `null`), while timestamps nested inside `Array`/`Map`/`Tuple` columns are rendered in the column's declared zone, else the ClickHouse server's, as the driver returns them — byte-identical to the [SSE stream](#get-v1stream--server-sent-events-stream) for values [canonicalized at ingest](#timestamp-canonicalization) (a fail-open pass-through that ClickHouse accepted still comes back canonical here, though it streamed in the producer's spelling). The response carries an `X-Cache: HIT` or `X-Cache: MISS` header — this endpoint shares the query cache + singleflight machinery (unlike `/v1/ops/query`, which always hits ClickHouse), keyed by [tenant](/deployment#multi-tenant-deployments): a request is never served from, or coalesced with, another tenant's. +JSON array of result rows, **rendered by ClickHouse**: the query runs over its HTTP interface with `FORMAT JSONEachRow`, and WaveHouse frames the lines into an array without re-encoding a value. So every type is spelled the way the connected server spells it, per version, with no WaveHouse conversion table in between. + +| ClickHouse type | JSON | +| --- | --- | +| `DateTime`, `DateTime64` | `"2026-06-21 04:00:00.123"` — space-separated, no `Z`, in the column's declared zone else the server's; byte-identical to the [SSE stream](#get-v1stream--server-sent-events-stream) for the same row (see [Timestamp rendering](#timestamp-rendering)) | +| `Decimal*` | a JSON **number** (`12.5`), not a string | +| `Int64`/`UInt64` past 2^53 | an unquoted number — still lossy in a JavaScript `number`; read it as text if you need every digit | +| `FixedString(n)` | a string padded to `n` bytes with `\u0000` | +| `Enum*` | the name, not the ordinal | +| `Nullable(T)` | `null` for a SQL `NULL` | +| `Array`, `Map`, `Tuple` | ClickHouse's own JSON for the container | +| `NaN` / `Inf` | `null` (ClickHouse's default rendering) | + +Keys come back in **SELECT order**, not alphabetical. The query runs with `readonly=2` and a server-side `max_execution_time`, so a statement that writes cannot slip through a read path and a runaway query is stopped by ClickHouse rather than only abandoned by WaveHouse. A filter value of `null` is a `400`, and an `in` list travels as one `Array(String)` parameter with a size check that answers `400` before the request is sent. The response carries `X-Cache: HIT` or `X-Cache: MISS` — this endpoint shares the query cache + singleflight machinery (unlike `/v1/ops/query`, which always hits ClickHouse), keyed by [tenant](/deployment#multi-tenant-deployments): a request is never served from, or coalesced with, another tenant's. The cache stores ClickHouse's own bytes. The inbound request body is capped at 1 MiB; a body over the cap is rejected with `413`. A query AST is bounded by nature (far under 1 MiB even with a large `in`-list), and the cap blocks a single-request memory-exhaustion vector on this public endpoint. Set a tighter or higher outer limit at your [reverse proxy](/reverse-proxy#request-body-size-limits) — but it can only narrow the effective limit, not raise it past this cap. @@ -591,7 +572,9 @@ The inbound request body is capped at 1 MiB; a body over the cap is rejected wit | Status | Body | Cause | | ------ | ---- | ----- | -| 400 | `{"error":"..."}` | Schema validation error (unknown column, bad aggregation, or an unparseable `time_range` `since`/`until` — neither a relative duration nor an RFC3339 timestamp) | +| 400 | `{"error":"unknown column: x"}` | Schema validation error — an unknown column, a bad aggregation, or an unparseable `time_range` `since`/`until` (neither a relative duration nor an RFC3339 timestamp) | +| 400 | `{"error":"filter value must not be null"}` | A filter carries `"value": null`. It is refused rather than answered: `col = NULL` is never true, and an empty parameter would silently ask a different question | +| 500 | `{"error":"clickhouse query: Code: 158. DB::Exception: … (TOO_MANY_ROWS) …"}` | ClickHouse refused the query — a resource limit, a type mismatch, anything else its engine raises. The body carries ClickHouse's own wording | | 403 | `{"error":"forbidden"}` | Role lacks select permission on table | | 403 | `{"error":"column \"x\" not allowed"}` | Column denied by policy | | 403 | `{"error":"aggregation \"x\" not allowed"}` | Aggregation fn denied by policy | @@ -662,26 +645,26 @@ Opens a persistent SSE connection for real-time event streaming. Supports histor **Response:** SSE stream (`text/event-stream`). Data events include an `id:` field set to the event's `received_timestamp`. The stream opens with a `: connected` comment and emits a minimal `:` keepalive comment periodically (every 30 seconds by default), which keeps a quiet connection from being closed by a proxy; both are standard SSE comments that `EventSource` ignores (raw consumers should skip `:`-prefixed lines). When the server stops (see [Stopping](/deployment#stopping)) it ends every open stream immediately rather than holding it for the drain; `EventSource` reconnects on its own and resumes from `Last-Event-ID`. A reload that stops serving the stream's tenant — its folder removed or rejected, over a [nested settings directory](/deployment#the-nested-settings-directory) — ends that tenant's open streams the same way, and the reconnect then gets its `404` (removed) or `503` (rejected): the SDK stops on the `404` and retries the `503`, resuming from `Last-Event-ID` once the folder is back, while a browser `EventSource` treats either as fatal. A browser going cross-origin reads either refusal only when it passes CORS: it is decorated from tenant `0`'s list ([multi-tenant deployments](/deployment#multi-tenant-deployments)), so where tenant `0` is not served or its list does not admit the page's origin, the SDK sees a network error instead and keeps re-dialing. -**Row values arrive positionally, and the column names are announced separately.** Before the first row, and again whenever the column list changes, the stream sends an `event: schema` frame naming the columns of the rows that follow — in order, already reduced to what the caller's role may read. That re-announcement is **not** guaranteed after a gap-fill across a column change; see the arity note below. Every data frame's `row` array then has exactly one value per announced column, in that order. `schema` is a **named** SSE event, so a browser `EventSource` must `addEventListener('schema', …)` — it never reaches `onmessage`. A schema frame carries **no** `id:` line, so it never moves the client's `Last-Event-ID`. In the example below the table has its own `received_timestamp` **column**, which collides by name with the frame's top-level `received_timestamp` **field** — they are different values: the field is when WaveHouse received the event, the row slot is that column as published (`null` where the record omitted it, which ClickHouse replaces with the column's default on insert). +**Row values arrive positionally, and the column names are announced separately.** Before the first row, and again whenever the column list changes, the stream sends an `event: schema` frame naming the columns of the rows that follow — in order, already reduced to what the caller's role may read. That re-announcement is **not** guaranteed after a gap-fill across a column change; see the arity note below. Every data frame's `row` array then has exactly one value per announced column, in that order. `schema` is a **named** SSE event, so a browser `EventSource` must `addEventListener('schema', …)` — it never reaches `onmessage`. A schema frame carries **no** `id:` line, so it never moves the client's `Last-Event-ID`. In the example below the table has its own `received_timestamp` **column**, which collides by name with the frame's top-level `received_timestamp` **field** — they are different values: the field is when WaveHouse received the event (WaveHouse's own RFC 3339 timestamp), the row slot is that column as ClickHouse rendered it (a record that omitted it carries the evaluated `DEFAULT`, not `null` — see [Timestamp rendering](#timestamp-rendering)). ```text event: schema data: {"table_name":"clicks","columns":["page","button","score","received_timestamp"]} id: 2026-03-24T12:00:00.123Z -data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:00.123Z","row":["/home","signup",42.5,"2026-03-24T11:59:58.512Z"]} +data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:00.123Z","row":["/home","signup",42.5,"2026-03-24 11:59:58.512"]} id: 2026-03-24T12:00:01.456Z -data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,null]} +data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,"2026-03-24 12:00:01.456"]} ``` A raw consumer must keep the most recent announced column list and zip each `row` against it; a value the record did not carry arrives as `null` in its slot rather than being omitted, so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you and still yields row objects — `.stream()` and `.liveQuery()` are unchanged. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. Each SSE connection is bound to a single `?table=`; to consume multiple tables, open one connection per table. -Values of top-level `DateTime`/`DateTime64` columns inside `row` arrive in the canonical RFC 3339 UTC form (ingest rewrites them before publishing — see [timestamp canonicalization](#timestamp-canonicalization)), so a live event and a `/v1/query` read of the same row agree on the instant in zone-explicit form — a zone-less spelling no longer parses as local time in a browser ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). The two renderings are byte-identical regardless of the declared time zone or a `Nullable` wrapper — a column declared with a non-UTC zone also streams as `Z`, and `/v1/query` normalizes it (nullable or not) to UTC before rendering. Canonicalization is fail-open at ingest, so a value outside the accepted input forms streams in whatever spelling the producer sent — and for exactly those events the byte-identity above does not hold: a spelling ClickHouse accepts anyway is stored and still queries back canonical, while one it too rejects lands in the DLQ and never becomes queryable at all. +Values of top-level `DateTime`/`DateTime64` columns inside `row` are ClickHouse's own rendering of the stored value — the exact bytes chtypes' `RowsExport` produced for that record (see [Timestamp rendering](#timestamp-rendering)), not a WaveHouse rewrite — so a live event and a `/v1/query` read of the same row agree on spelling **by construction**, with no separate canonicalization step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). A column declared with a non-UTC zone streams in that zone, not normalized to UTC; parse the timestamp with a zone-aware parser rather than assuming `Z`. -**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. For a filter constant the query path's SQL also accepts ([the enforcement caution](/access-control#where-each-rule-is-enforced) gives per-type guidance), a connection is never delivered a row the query path would hide for that role — every comparison the stream can't prove fails closed and withholds the row instead. Numeric comparisons run in the column's storage domain — both operands narrowed the way ClickHouse narrows the stored value and the bound constant — so columns that narrow on insert (`Float32`/`Float64` width, a `Decimal`'s scale) agree with the query path too; the residual payload-vs-stored case is an event whose insert later fails into the DLQ, which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), but an expired token or changed claims take effect only when the client reconnects. +**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause — a connection is never delivered a row the query path would hide for that role, and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: its rows are withheld with reason `unavailable` while other tenants' streams are unaffected. The residual payload-vs-stored case is an event whose insert later fails outright at ClickHouse — a connectivity fault or batch error, not a data-shape problem chtypes would already have caught — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next live event (an in-flight gap-fill finishes under the policy snapshot taken when the stream opened), but an expired token or changed claims take effect only when the client reconnects. **CORS:** `/v1/stream` honors the request's tenant's `cors.allowed_origins` allowlist (settings directory) like every endpoint — the preflight included, which a browser sends without `X-Tenant-ID`, so over [a nested settings directory](/deployment#multi-tenant-deployments) the fronting proxy has to set the header on the `OPTIONS` too. Note that a **header-authenticated stream preflights before it connects** — `Authorization` is not CORS-safelisted — where a bare `EventSource` never preflighted at all: its request is not a `fetch()`, so Fetch's unsafe-request flag is never set and `Last-Event-ID` rides on the plain `GET`. Both headers are allow-listed, so an allowed origin connects *and* resumes cross-origin. @@ -894,18 +877,20 @@ The message format used on NATS JetStream between ingest and the batch consumer: "received_timestamp": "2026-03-24T12:00:00.123456789Z", "format": "JSONCompactEachRow", "columns": ["page", "button", "score", "received_timestamp"], - "row": ["/home", "signup", 42.5, null] + "row": ["/home", "signup", 42.5, "2026-03-24 12:00:00.123"] } ``` +The request that produced this envelope omitted `received_timestamp` (`DEFAULT now64(3, 'UTC')` on the table); chtypes evaluated the default before publish, so `row` carries the resulting timestamp — ClickHouse's own rendering, not `null` and not a WaveHouse rewrite. + | Field | Type | Description | | ----- | ---- | ----------- | | `table_name` | string | Target ClickHouse table (from URL). | | `scope` | string | Reserved; currently always empty. | | `received_timestamp` | string | RFC 3339 nano timestamp when WaveHouse received the event. | | `format` | string | Row format. Always `JSONCompactEachRow` today; stated on the wire so a reader can tell an envelope it understands from one it doesn't. | -| `columns` | string[] | The table's **insertable** column names, in declaration order — what each position in `row` means. A `MATERIALIZED` or `ALIAS` column is computed by ClickHouse and cannot be named in an `INSERT`, so it never appears here. | -| `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order. A column the request body omitted is `null` here; for a **non-nullable** column the insert turns that back into the column's default (`input_format_null_as_default`), but a `Nullable(T) DEFAULT …` column stores `NULL` — only an *absent* key ever took the default, and a positional row has one slot per column and no way to express absence. Parseable `DateTime`/`DateTime64` values are rewritten to canonical RFC 3339 UTC (see [timestamp canonicalization](#timestamp-canonicalization)); other values as originally sent. | +| `columns` | string[] | The table's **wire** column names, in declaration order — what each position in `row` means (`internal/typelayer.Table.WireColumns`: the insertable subset minus any `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column — none of the three can be named in an `INSERT`, or is ever part of a published row). | +| `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order — the exact bytes ClickHouse's own writer produced for this stored row (`internal/typelayer.Table.Ingest`, via chtypes). A column the request body omitted carries its evaluated `DEFAULT` (or the type's implicit zero value where none is declared), not `null` — the same as a native `INSERT` naming fewer columns than the table has. `DateTime`/`DateTime64` values are ClickHouse's own rendering (see [Timestamp rendering](#timestamp-rendering)), and an out-of-range integer is wrapped the way a real `INSERT` wraps it. | `columns` and `row` are only meaningful together: a reader that cannot pair them — a length mismatch, an undecodable row, a `columns` list naming one column twice — has no way to map a value to a column. Both readers also refuse an envelope whose `format` they do not recognize. Either way the SSE fan-out withholds such an envelope rather than guess, and the batch consumer parks it on the DLQ with `X-DLQ-*` headers — acking and dropping it only where the DLQ is switched off for that table, since it can never insert on retry. Both outcomes increment `wavehouse_ingest_poison_total`, separated by its `disposition` label (`parked` / `dropped`). @@ -925,7 +910,7 @@ Three values, where the envelope above has four: this is the frame a role restri ## Dead Letter Queue (DLQ) -When ClickHouse **rejects** a batch insert (a value it cannot parse, a type mismatch, a table or column it does not have), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A ClickHouse that **cannot take** the insert — down, unreachable, timing out, overloaded, read-only, or refusing WaveHouse's credentials — never sends a row here: the batch stays in the tenant's ingest queue and is retried with backoff until it inserts (see [Ingest Pipeline](/ingest-pipeline#when-clickhouse-cannot-take-an-insert)). A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: malformed JSON, an envelope of an unknown `format`, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. +When ClickHouse **rejects** a batch insert (a value it cannot parse, a type mismatch, a table or column it does not have), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A ClickHouse that **cannot take** the insert — down, unreachable, timing out, overloaded, read-only, or refusing WaveHouse's credentials — never sends a row here: the batch stays in the tenant's ingest queue and is retried with backoff until it inserts (see [Ingest Pipeline](/ingest-pipeline#when-clickhouse-cannot-take-an-insert)). A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: malformed JSON, an envelope of an unknown `format`, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values exactly as published: ClickHouse's own rendering of the stored value, since chtypes already validated and coerced the record before it was ever published — see [Timestamp rendering](#timestamp-rendering)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. Because chtypes catches the type and shape problems synchronously at ingest, a row that reaches this DLQ path failed for a reason chtypes couldn't have caught up front — a ClickHouse-side outage or a genuine insert-time fault — not a data mismatch. Under [`mq.backend: nats`](/deployment#external-nats) the parked rows of every tenant go to one shared dead-letter stream instead, under `.dlq.{tenant}.{table}`; the bodies and headers are the same. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 94ead446..7945b050 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -45,7 +45,7 @@ flowchart TD ## Binaries -WaveHouse ships a single binary, `wavehouse`: an all-in-one process running the API, batch worker, embedded NATS JetStream, and optional embedded Pebble dedup. The only external dependency is ClickHouse, unless a shared backend is selected: `cache.backend: redis`, `dedupe.backend: dynamodb`, or `mq.backend: nats`, which points the queue at a NATS cluster the operator runs and lets several processes, each running some of the [roles](/configuration#process-roles), share it. `cmd/wavehouse` is the shell — subcommand dispatch, the logger, `config.Load`, the signal context — and `internal/app` is the process itself (see [`app/`](#app--process-wiring) below). +WaveHouse ships a single binary, `wavehouse`: an all-in-one process running the API, batch worker, embedded NATS JetStream, and optional embedded Pebble dedup. The binary also loads a second artifact at start: a per-ClickHouse-version shared library (`internal/typelayer`, via [chtypes](/deployment#chtypes-artifacts)) that runs ClickHouse's own parser in-process for ingest validation, type coercion, and row-level security, and only a process running the `api` role loads it. The binary requires cgo (dlopen only — no static link to the artifact) and glibc, so supported platforms are Linux amd64/arm64 and macOS arm64. The only external network dependency is ClickHouse, unless a shared backend is selected: `cache.backend: redis`, `dedupe.backend: dynamodb`, or `mq.backend: nats`, which points the queue at a NATS cluster the operator runs and lets several processes, each running some of the [roles](/configuration#process-roles), share it. `cmd/wavehouse` is the shell — subcommand dispatch, the logger, `config.Load`, the signal context — and `internal/app` is the process itself (see [`app/`](#app--process-wiring) below). ## Internal Packages @@ -56,21 +56,22 @@ internal/ ├── auth/ JWT/JWKS authentication middleware (HMAC or JWKS, role extraction) ├── cache/ Query cache: Ristretto L1 + the tenant-led version index; the Redis-compatible shared backend ├── chconn/ One ClickHouse pool per connection tuple among the served tenants, reconciled on reload under the ceiling -├── chsql/ Shared ClickHouse SQL helpers (identifier quoting, bind-safety) +├── chsql/ Shared ClickHouse SQL helpers (identifier quoting, bind-safety, strict integer cast) ├── config/ YAML + env var configuration loading ├── coord/ Leases for work that must run in one process at a time (the sweeper, the ingest processes' shard claim membership), with fencing tokens (the NATS KV implementation is internal/mq/lease.go) ├── dedupe/ Optional deduplication (Reserve/Commit/Release; Pebble, DynamoDB) -├── discovery/ ClickHouse schema introspection and validation +├── discovery/ ClickHouse schema introspection (system.columns/system.tables, server version + timezone) ├── ingest/ Batch buffering, DLQ, Active Sweeper, and shard claims over a shared queue ├── keyenc/ The one escaping composite keys are built from (NATS subject tokens, cache keys, dedupe keys) ├── mq/ MQ boundary: the only NATS/JetStream importer (owned message/consumer/stream types + embedded server) ├── observability/ OpenTelemetry pipeline (traces/metrics/logs + Prometheus exposition) ├── pipes/ Named query pipes (NamedQuery type, parameter binding, Source) -├── policy/ Hasura-style access control (policy types, evaluation, Source) +├── policy/ Hasura-style access control (policy types, claim resolution, Source) ├── query/ Structured query AST, SQL builder, and timestamp bucketing ├── settings/ Settings directory: validate the JSON files, hold each tenant's adopted snapshot, reload on watch / SIGHUP / API ├── stream/ SSE fan-out: event Hub (project once per role), Subscriber queue, Bucket fan-out, keepalive Heartbeater wheel -└── tenant/ Tenant id: the type, its grammar, the reserved default, the request header name +├── tenant/ Tenant id: the type, its grammar, the reserved default, the request header name +└── typelayer/ In-process ClickHouse parser (chtypes): one engine per process, a table set per tenant — ingest validation/coercion + predicate compilation ``` ### `api/` — HTTP Layer @@ -83,8 +84,9 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`clickhouse_exec.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. -- **ingest.go** — Accepts `POST /v1/ingest?table={table}` in three body shapes: one flat JSON object, a JSON array of them, or NDJSON. The **required** `Content-Type` chooses the format *family* — `application/json` versus the four NDJSON spellings — and within the JSON family the body's first non-whitespace byte picks array versus single object; the bytes never choose the family. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type — a comma inside a *quoted* parameter value is data, so `application/json; a=", application/x-ndjson; b="` is accepted. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer and runs the per-format record readers over those bytes, so the `413` lands before any record is processed and peak memory per request is O(body) rather than O(record). Then it validates and encodes each record, and runs the records in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row missing the configured `id_field` (or setting it to `null`) can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). -- **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` (the Go-side type conversion lives in the structured-query / pipes path, not here). +- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch, and the dedupe id is read positionally out of the exported row. One `typelayer.Table.Ingest` call per body, on the table set bound for the request's tenant, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5`, decided before the body is read and ahead of every other tenant's traffic. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent or `null` can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). +- **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` — a deliberately different audience from the structured-query path, which leaves ClickHouse's default spelling alone so it matches the SSE wire. +- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every filter value bound as a named `{pN:String}` parameter, so ClickHouse renders each row and WaveHouse only frames the lines into an array. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=simple`) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. Connections per tenant are capped at the tenant's `max_open_conns`. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) driver.Conn` beside it — each resolves the request's tenant per call, and a nil connection (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for one tenant's per-table parked counts (optionally one table) and its total — the tenant `?tenant=` names, read strictly by `opsTenant`, tenant `0` without it. The tenant is looked up in the MQ, not the settings registry, so a rejected or removed tenant's parked rows are read like a served one's; a tenant with no dead-letter queue (`mq.ErrNoDeadLetterQueue`) is a 404, and any other failure to read it a 500. The queue itself is `internal/mq`'s. @@ -103,11 +105,11 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)) lives next to the keepalive primitives it shares. One abstraction per file. -- **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row (`ResolvedPermissions.RowVisible`, evaluated against the full event via the type-aware comparison seeded from the schema registry — `policy.ColumnSpec`: numeric columns compare numerically, `String` bytewise, `DateTime`/`DateTime64` as instants through the same parser ingest canonicalization uses (`discovery.Column.TimeParser` — one grammar, so filter constants and canonicalized payloads can't disagree on the instant), everything else admits byte-equality only and fails ordering/`!=` closed, so a missing schema can never downgrade the comparison to a leak). Each row withheld this way increments `wavehouse_sse_rows_withheld_total`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and caching the per-table column-kind lookup across the replay loop. `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. +- **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row. Visibility itself is decided by `internal/typelayer` (`Table.ParseRow` once per event, on the event's tenant's table set, then `Row.Visible` per subscriber) — the same ClickHouse parsing and comparison semantics the server's own `WHERE` clause applies, for every column type, rather than a hand-written per-type comparator; a predicate error, a policy column the table no longer has, a schema drift between the event and the live table, or an engine that is unavailable for that tenant all withhold the row rather than guessing. Each row withheld this way increments `wavehouse_sse_rows_withheld_total{table,role,reason}`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and caching the per-table column-kind lookup across the replay loop. `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. - **subscriber.go** — `Subscriber`, the per-connection handle. It carries the connection's JWT claims, fixed at construction (`NewSubscriber(claims, metrics)`, no setter) — the claims the `Hub` resolves a role's row-level `filter` against, and immutability is what makes the fan-out's unsynchronized claims read race-free structurally. It owns a single ready-to-write outbound queue of `Frame`s (each tagged with its `kind`, so the handler labels the write where it happens): producers — the keepalive wheel and the event `Hub` — fan frames in with `Send` (non-blocking; a full queue drops, and `Send` itself counts the drop by frame kind, so no producer can forget to), and the handler drains `Frames()` to the client verbatim. The queue is sized for buffering live events (cap 64, up from the keepalive-only cap 1; #152 will make it a knob), and `Evict` closes its `Evicted()` channel, once, for the handler to end the stream: the `Hub`'s `Prune` does for a tenant no longer served, and the slow-consumer follow-up will for a wedged consumer. - **bucket.go** — `Bucket`, the reusable fan-out primitive: a concurrency-safe set of subscribers. `Push` fans one `Frame` to every member fire-and-forget — the keepalive wheel's ring is its only caller now that both `Hub` paths iterate `Snapshot`, since the schema announcement is per connection even where the projection is shared per role; `Snapshot` exposes the members so the event `Hub` can evaluate row visibility per subscriber before sending (drop counting lives in `Send` itself). The `Hub` holds one `Bucket` per `(topic, role)` so a projected frame is built once and sent to every member instead of re-projected per subscriber. - **heartbeat.go** — The keepalive wheel (`Heartbeater`). A single process-wide ticker fans a minimal `:` comment across the ring of `Bucket`s, waking ~1/N of live streams per tick so the writes don't synchronize. The effective per-connection keepalive period is `stream.keepalive_interval` in the settings directory (the wheel ticks every `keepalive_interval ÷ keepalive_buckets`, so one rotation spans the interval; a reload calls `Reconfigure`, which rebuilds the ring in place with every live subscriber carried over); the owning handler goroutine does the actual write, so the shared ticker never touches a `ResponseWriter` directly. -- **metrics.go** — `Metrics`, the SSE instrument set: `wavehouse_sse_active_streams` (open streams), `wavehouse_sse_stream_duration_seconds` (lifetime), `wavehouse_sse_frames_sent_total` / `wavehouse_sse_bytes_sent_total` (labeled by `kind`: `keepalive`, `event`, `replay`, `schema`), `wavehouse_sse_dropped_frames_total` (frames dropped to a full subscriber queue — the slow-consumer signal that was silent before #294), and `wavehouse_sse_rows_withheld_total` (rows withheld from a subscriber by the row-level-security filter, labeled by table and role — the signal that separates "no matching rows" from "a fail-closed filter is withholding everything"). Nil-safe, so the handler holds one unconditionally and tests skip wiring it; one shared instance records the handler's write sites, each `Subscriber`'s queue-full drops (counted inside `Send`, by frame kind), and the `Hub`'s row-withheld counts. Separate from `observability.RegisterSystemMetrics`, which covers only the NATS/Pebble system gauges. Streams are observed through these metrics rather than per-event traces (the router excludes `/v1/stream` from the HTTP tracer). +- **metrics.go** — `Metrics`, the SSE instrument set: `wavehouse_sse_active_streams` (open streams), `wavehouse_sse_stream_duration_seconds` (lifetime), `wavehouse_sse_frames_sent_total` / `wavehouse_sse_bytes_sent_total` (labeled by `kind`: `keepalive`, `event`, `replay`, `schema`), `wavehouse_sse_dropped_frames_total` (frames dropped to a full subscriber queue — the slow-consumer signal that was silent before #294), and `wavehouse_sse_rows_withheld_total` (rows withheld from a subscriber by row-level security, labeled by `table`, `role`, and `reason` — `filter` (a definite non-match) vs. `error`/`decline`/`unavailable`/`drift` — the signal that separates "no matching rows" from "a fail-closed filter is withholding everything"). Nil-safe, so the handler holds one unconditionally and tests skip wiring it; one shared instance records the handler's write sites, each `Subscriber`'s queue-full drops (counted inside `Send`, by frame kind), and the `Hub`'s row-withheld counts. Separate from `observability.RegisterSystemMetrics`, which covers only the NATS/Pebble system gauges. Streams are observed through these metrics rather than per-event traces (the router excludes `/v1/stream` from the HTTP tracer). ### `auth/` — Authentication @@ -154,19 +156,30 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ - **dedupetest/** — the conformance suite every backend runs: `Run(t, newHarness)` drives the contract above through a backend's `Factory` (claim, commit, release, lease lapse, retention, one claim among concurrent reserves from two clients, keyspaces, input order, late commit, stale release, failure mid-call); `Harness` optionally injects a clock and a mid-call failure. `Mark` is the old check-and-mark in one call, for tests that only need an id seen. - **stores.go** — `Stores` is one `Managed` per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7), built on first use through a `Factory` (`func(tenant.ID) *Managed`) — whether tenants share a backend is the factory's business (`Embedded.Tenant` puts them all in one Pebble instance), with nothing that holds the `Stores` changing. `For(id)` returns a tenant's store, built closed so a tenant adopted a moment ago answers `ErrDisabled` rather than having no store; `Retain(keep)` closes and forgets the stores of tenants no longer served, touching nothing on disk; `Close()` closes every store. `Factory.Gated(ready)` wraps a factory so a store opens only once `ready` returns nil, and fails closed until then (the DynamoDB wiring's table check). `internal/app` drives it from the registry's `AfterAdopt` hook. -### `discovery/` — Schema Discovery & Validation +### `discovery/` — Schema Discovery -- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant to another user or `tls` block, which reads the same tables, applies to the next refresh, one that moves it to another address or database has `internal/app`'s `discoveries` drop its registry and build a fresh one, whose first discovery runs at once and whose lookups answer `ErrNotLoaded` until it succeeds ([#638](https://github.com/Wave-RF/WaveHouse/issues/638)), and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) can decide the insertable subset the ingest envelope and the SSE announcement are both built from. Each refresh also records the server version (`SELECT version()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), reads each column's `default_expression` and 1-based `position` alongside its type, discovers the server's default time zone (`SELECT timezone()`) and bakes every `DateTime`/`DateTime64` column's canonicalization spec (precision + resolved zone) into the cached schema, so the per-record ingest path parses no type strings and loads no zones ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop, each sleep drawn uniformly below the backoff so instances retrying one ClickHouse do not retry in lockstep, used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. -- **timestamp.go** — `CanonicalizeTimestamps(schema, data)` rewrites every parseable value in a top-level `DateTime`/`DateTime64` column to the canonical RFC 3339 UTC wire form before the event is published ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)): zone-less values are interpreted in the column's declared zone, else the discovered server default — ClickHouse's own rule, so the spelling changes but never the instant. Fail-open: an unparseable value or unresolvable zone passes through verbatim for ClickHouse's own parser to judge; ingest never rejects a record over its timestamp spelling. `Column.TimeParser()` exposes the same grammar as a value→instant parser (nil for a column with no resolved timestamp spec — a non-timestamp column, or one whose declared zone couldn't be loaded), which the stream row-filter uses so filter constants and canonicalized payloads can't disagree on the instant ([#381](https://github.com/Wave-RF/WaveHouse/issues/381)). -- **validation.go** — `Validate(schema, data)` checks incoming JSON against the discovered schema: unknown fields, type compatibility, missing required columns, null handling. Also exports the type classifiers `IsNumericType` / `IsStringType` and the storage-model classifier `NumericStorageOf` (all unwrapping `Nullable`/`LowCardinality`; the latter yields a numeric column's float width, `Decimal` scale, or integer exactness), which — together with `Column.TimeParser` from timestamp.go — seed the stream row-filter's `policy.ColumnSpec` comparison. -- **discovery_test.go** — Unit tests for validation logic. +- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant to another user or `tls` block, which reads the same tables, applies to the next refresh, one that moves it to another address or database has `internal/app`'s `discoveries` drop its registry and build a fresh one, whose first discovery runs at once and whose lookups answer `ErrNotLoaded` until it succeeds ([#638](https://github.com/Wave-RF/WaveHouse/issues/638)), and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) can decide the insertable subset the ingest envelope and the SSE announcement are both built from. Each refresh also records the server version (`SELECT version()`) and default time zone (`SELECT timezone()`, exposed via `ServerTimezone()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), and reads each column's `default_expression` and 1-based `position` alongside its type. An `OnRefresh` hook fires synchronously after every successful refresh and **before** the registry reports itself loaded, so a loaded tenant is a bound one — `internal/typelayer.Engine.Bind` is its only registered consumer, and it is what resolves the chtypes artifact matching that tenant's server line and recompiles its per-table handles ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop, each sleep drawn uniformly below the backoff so instances retrying one ClickHouse do not retry in lockstep, used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. +- **discovery_test.go** — Unit tests for schema discovery. + +### `typelayer/` — In-Process ClickHouse Parser + +The only package that imports `github.com/wave-rf/chtypes/go/chtypes` (the sole exception: `cmd/wavehouse/main.go` references `typelayer` itself, never chtypes directly). One process-wide `Engine` wraps one lazily-opened `chtypes.Registry`, found through a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`; empty means the chtypes search path), and only a process running the `api` role builds it — an ingest-worker-only or sweeper-only process never loads the artifact and boots without one. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for what ships where and how large it is. + +Each tenant has its own table set inside that engine, bound from the tenant's own schema refresh and released when the tenant is no longer served. + +- **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind. A tenant whose ClickHouse line has no installed artifact is unavailable on its own, and so is one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Either way the cause is named, every other tenant keeps working, and a later bind of the same tenant (a reload, a refresh that now agrees) clears it. +- **`Engine.RoleTable`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column is simply not in the schema, so naming it is ClickHouse's code 117, and an absent check column takes the claim as its default while a supplied value still wins. +- **`Table.Ingest(format, body)`** runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. +- **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `Table.Ingest` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow` / `Row.Visible` judge a subscriber's row filter over one parsed event, with compiled filters cached per table. Only a definite true admits; a predicate error, a policy column the table no longer has, schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. +- A tenant that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row for that tenant's tables with reason `unavailable`. + +See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest error-response shape and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced) for how predicates are compiled and evaluated. ### `ingest/` — Ingest Pipeline, DLQ & Sweeping -- **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line, with `columns` naming its positions (the table's insertable columns — a computed one cannot be named in an `INSERT`); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow`. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so under `mq.backend: embedded` the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Under `mq.backend: nats`, anyone the operator lets publish to `.ingest.>` reaches it too, past auth, policy and schema validation, so that right belongs to the `wavehouse` user alone. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy — or through an operator-authored [pipe that writes](/pipes#pipes-that-write), gated only by its `allowed_roles`. A batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling) is never tried: no row of it could pass, so `parkBatch` takes it to the DLQ switch whole, logging once per batch rather than twice per row. Otherwise a bulk-insert failure is first classed by `chconn.Classify`: a ClickHouse that cannot take the insert (unavailable, denied, or no verdict at all) sends the batch back to the MQ for a delayed redelivery (`retryLater` → `mq.Message.NakWithDelay`), under a backoff shared by every table on the same pool (a failure of one table — read-only, too many parts — backs off that table alone), and never to the DLQ — the same when it stops answering mid-isolation. Only when ClickHouse rejects the batch, or refuses a multi-row batch for its size (`chconn.Splittable`: too many partitions for one INSERT, the memory limit), is it re-inserted row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. +- **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line — the exact bytes ClickHouse's own writer produced for the stored record — with `columns` naming its positions (the table's insertable columns, or the narrower list a column-restricted role produced); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with `typelayer.InsertSettings()` (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) plus `async_insert=0`, the same parsing settings chtypes compiled the row with. The worker never loads the artifact: `InsertSettings` is static. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so under `mq.backend: embedded` the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Under `mq.backend: nats`, anyone the operator lets publish to `.ingest.>` reaches it too, past auth, policy and schema validation, so that right belongs to the `wavehouse` user alone. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy — or through an operator-authored [pipe that writes](/pipes#pipes-that-write), gated only by its `allowed_roles`. A batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling) is never tried: no row of it could pass, so `parkBatch` takes it to the DLQ switch whole, logging once per batch rather than twice per row. Otherwise a bulk-insert failure is first classed by `chconn.Classify`: a ClickHouse that cannot take the insert (unavailable, denied, or no verdict at all) sends the batch back to the MQ for a delayed redelivery (`retryLater` → `mq.Message.NakWithDelay`), under a backoff shared by every table on the same pool (a failure of one table — read-only, too many parts — backs off that table alone), and never to the DLQ — the same when it stops answering mid-isolation. Only when ClickHouse rejects the batch, or refuses a multi-row batch for its size (`chconn.Splittable`: too many partitions for one INSERT, the memory limit), is it re-inserted row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. - **backoff.go** — The retry backoff behind `retryLater`: a small circuit breaker per ClickHouse pool (the target's URL, user and database), and one per pool and table for a failure of one table (`chconn.TableScoped`). A failure opens it for 1 s, doubling to a 30 s cap, each window jittered down to half; while it is open, flushes and arriving rows are handed back without a request, and once it elapses one flush probes. Any answer that is not an outage closes it. -- **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. -- **compact.go** — `EncodeCompactRow`, the positional row encoder every published row goes through, rendering one record over the table's **insertable** columns in declaration order. Serialization only: it validates nothing and judges no value. +- **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. `Row` is the exact bytes `internal/typelayer.Table.Ingest` returned for an accepted record — ClickHouse's own `JSONCompactEachRow` writer output, `DEFAULT`s already filled in — not a value WaveHouse encodes itself. - **claims.go**, **assign.go** — `ClaimShards` wraps a `Sharded` queue so an ingest process consumes only its share of the units, one process per unit at a time. Each process holds one membership lease, `ingest.m` for the lowest free `j` under the number of configured units (at most 64, read 16 at a time each tick), and every tick (2s) reads which slots are live (`coord.Observer.Held`); `assignUnits` gives every unit to a live slot by rendezvous hashing capped at ⌈units/slots⌉, the same result in every process for the same view; the extras are assigned apart from the configured units, so processes whose lists of extras differ (each refreshes it every five minutes) still agree on every configured unit. A unit no longer assigned halts (`mq.Halter`: stops fetching, keeps its pin, returns once what it fetched reached the worker), waits (bounded by the worker's 60s ack wait, the shortest `ack_wait` a durable may have) for the rows it delivered to settle (`Message.OnSettled`), then releases its pin (`mq.Releaser`). `Halt` on the claiming consumer ends the ticks and halts every unit at once; the worker calls it before its final flush, and the claims' stop releases the units and resigns the membership lease after it. A unit taken over from a slot that is no longer live is reset (`ResetOrphaned`) before it is bound — waiting, unbound, while the broker answers `ErrUnitHeld` (the dead owner's pin has not lapsed, or the unit was active within its pinned TTL), up to 30s — so the dead owner's unacked rows come back at once; on a process's first tick only if it is the one live member (a restart after a clean stop; after a crash the dead run's lease still counts as live for a lease duration). Each unit's rows delivered and unsettled are capped at its share of the worker's 10,000 (`ClaimConfig.MaxHeld`): an even share over the units assigned (`unitShare`, recomputed each tick and read by the broker before every fetch through `ConsumerConfig.MaxHeld`), never under 1,000, two batches; at its share a unit fetches only what keeps its pin, and one stuck unit never takes another's. A row stops counting at its first ack or nak attempt (`Message.OnSettled` fires whether or not the broker confirms), or after `ack_wait` if the worker never settles it, since the broker redelivers it then as a new row. A configured unit whose delivery ends, or whose durable or stream is gone or whose durable no longer fits (`mq.ErrConsumerNotFound`, `mq.ErrConsumerMismatch`) when it is bound, fails the worker; any other bind failure is retried next tick, logged as an error once it has lasted a minute, and an extra's end is logged. The lowest live slot counts the units with rows and no owner (`Sharded.Unowned`) for `wavehouse_ingest_shards_unowned`. - **sweeper.go** — `Sweeper` implements the Active Sweeper pattern. It runs every minute and asks the MQ (`mq.Purger.PurgeAcked`) to drop the ingest events that are **both** ACKed by the buffer consumer (written to ClickHouse) **and** older than the gap window (re-read every sweep: each tenant's own `stream.gap_window_minutes`, a rejected tenant's as its folder last had it (unbounded for one rejected since boot) — `internal/app`'s `gapWindows` — and none for a removed tenant). Finding the purge point is `internal/mq`'s (`purge.go`). @@ -197,12 +210,12 @@ The package's design invariants — stdout always 100%, WARN+ERROR always export ### `policy/` — Access Control -- **policy.go** — Hasura-style policy types, now **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant carries a separate `SelectPermissions` and `InsertPermissions` — so a field only one side ever honored (`filter`, aggregations and the `max_*` limits on select; `check` on insert) does not exist on the other, and a document that puts one there fails the strict decode as an unknown key. `Evaluate()` resolves permissions against JWT claims (including `{{ jwt.claim.path }}` template resolution) for ONE operation, leaving the side it did not resolve **nil**. That is what the pointers buy: an *empty* side means "no restrictions" — what the admin return constructs on both sides — while a *nil* side means "you asked the wrong operation", and as value types the two were the same zero value. Every accessor fails closed on a nil side. The handful of bare field reads outside this package each sit past an accessor that denies an unresolved side first, so a nil `Select` never reaches one; if that ordering ever changed they would panic rather than silently widen. A nil guard that skips such a read must never be added, since an absent `WhereClause` is an unfiltered query. The per-column decision `IsColumnAllowed(col, insert)` takes the side it is being asked about, alongside its batch/projection forms `AllowedProjection()` and `RestrictsColumns()`, `IsAggregationAllowed()`, `CheckClauses()` — the write-side accessor for the one consumer that iterates a side's map instead of asking about a column, whose `ok=false` a caller must treat as *refuse the write*, never as *no checks to run* — `resolvePredicates()` — the one resolution both read surfaces render from, so the SQL `WHERE` and the in-memory row check cannot drift — and `Validate()`, split into `validateSelectPerms`/`validateInsertPerms` and run from `settings.Validate` on every adoption, which is where the rules in [Access Control](/access-control) are actually enforced. -- **rowfilter.go** — the in-memory row-visibility twin of the SQL `WHERE`: `HasRowFilter`, `RowVisible` (evaluates the resolved predicates against a decoded event, per subscriber), and `ColumnSpec` — the per-column comparison contract (`ColumnKind` `Numeric`/`Text`/`Time`/`Opaque`, plus each kind's parameters: the caller-supplied instant parser for `Time`, the `NumericSpec` storage model for `Numeric`) whose zero value is the fail-closed floor: numeric columns compare in the column's **storage domain** (operands rendered by canonical.go, compared by numeric.go — next two bullets), `String` bytewise, `DateTime`/`DateTime64` chronologically (both operands through the ingest grammar; either side unreadable ⇒ withheld), and everything else (including any column with no usable schema) admits byte-equality only, failing `!=`/`>`/`<` closed. Both `HasRowFilter` and `RowVisible` fail closed on a denied or unresolved grant: `HasRowFilter` is the gate in front of `RowVisible`, so it must answer *true* there or the whole-bucket fast path skips the check entirely. -- **canonical.go** — the one rendering layer for comparison operands: every value a `filter` or `check` compares — a JWT claim (`CanonicalScalar`), a policy-authored literal (`CanonicalNumericLiteral`), an ingested payload value (`numericCanonical`) — converges on one exact canonical decimal form (positional, digit-bounded, never a float64 round-trip), so what a read filter binds and what the stream compares can't drift; `scalarString` is the deliberate exception, the raw byte rendering that `Text`/`Opaque` equality compares. -- **numeric.go** — compares canonical forms the way the column that stores them would: `compareCanonicalDecimals` orders by exact digit-string arithmetic, and `NumericSpec` first narrows both operands the way ClickHouse narrows the stored value and the bound constant — `Float32`/`Float64` width rounding, `Decimal` scale truncation, integers exact at any width, with an operand outside the column's width or a `Decimal`'s precision budget refused rather than modeled; the `tests/integration` differential oracle holds in-range verdicts equal to a live ClickHouse's and asserts the never-admit-where-SQL-hides direction for the refused out-of-range operands. +- **policy.go** — Hasura-style policy types, now **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant carries a separate `SelectPermissions` and `InsertPermissions` — so a field only one side ever honored (`filter`, aggregations and the `max_*` limits on select; `check` on insert) does not exist on the other, and a document that puts one there fails the strict decode as an unknown key. `Evaluate()` resolves permissions against JWT claims (including `{{ jwt.claim.path }}` template resolution) for ONE operation, leaving the side it did not resolve **nil**. That is what the pointers buy: an *empty* side means "no restrictions" — what the admin return constructs on both sides — while a *nil* side means "you asked the wrong operation", and as value types the two were the same zero value. Every accessor fails closed on a nil side. The handful of bare field reads outside this package each sit past an accessor that denies an unresolved side first, so a nil `Select` never reaches one; if that ordering ever changed they would panic rather than silently widen. A nil guard that skips such a read must never be added, since an absent `WhereClause` is an unfiltered query. The per-column decision `IsColumnAllowed(col, insert)` takes the side it is being asked about, alongside its batch/projection forms `AllowedProjection()` and `RestrictsColumns()`, `IsAggregationAllowed()`, `CheckClauses()` — the write-side accessor for the one consumer that iterates a side's map instead of asking about a column, whose `ok=false` a caller must treat as *refuse the write*, never as *no checks to run* — `resolvePredicates()` (exposed through `Predicates()`) — the one resolution both read surfaces render from, so the SQL `WHERE` and the chtypes filter the stream compiles cannot drift — and `Validate()`, split into `validateSelectPerms`/`validateInsertPerms` and run from `settings.Validate` on every adoption, which is where the rules in [Access Control](/access-control) are actually enforced. +- **canonical.go** — the one rendering layer for policy comparison operands: every JWT claim value (`CanonicalScalar`) is rendered into one exact canonical decimal form (positional, digit-bounded, never a float64 round-trip) before it reaches a `filter`/`check` predicate, so every comparison surface binds the same value the same way, and a null/object/array claim fails closed. A policy-authored literal is *not* re-rendered — it binds exactly as written, and a spelling the column cannot read is ClickHouse's own type error at evaluation time on both surfaces. On an integer column the bound claim is additionally compared through the strict round-trip cast (`WhereSQL(colType)` renders the predicate for the query builder; the stream's chtypes filter uses the same shape). - **source.go** — `Source`, a `func() *Policy` the `/v1/ops` gate (over a flat settings directory) reads per call, so a settings reload applies to the very next request; in production it is the default tenant's `settings.Store.Policy`, and `Static(p)` fixes one for tests. The tenant-aware surfaces take a keyed variant that resolves to the same `Store.Policy`: `api.PolicySource` (`func(*settings.Store) *policy.Policy`) for ingest, structured query and pipes, `stream.PolicySource` (`func(tenant.ID) *policy.Policy`) for the hub, and `auth.PolicySource` (same shape) for the operator key's admin role. A `nil` result is a deliberate lockout. +Predicate *evaluation* lives elsewhere: `internal/typelayer` compiles a role's resolved predicates through chtypes and evaluates them — `Table.ParseRow` / `Row.Visible` for a streamed event, `Table.Ingest` (a row filter inside the validating parse) for an ingest `check` — with ClickHouse's own comparison semantics for every column type. `policy` only resolves the values both the SQL path and that engine bind. + ### `pipes/` — Named Query Pipes - **pipes.go** — `NamedQuery` type with SQL template and parameter definitions, and `Source` (`Pipe(name)` / `Pipes()`), read per request — `settings.Store` in production (`pipes.json`), `Static(q...)` in tests. `BindParams()` resolves `{{param}}` / `{{param:default}}` placeholders by inlining escaped literal values into the SQL (strings single-quote-escaped; arrays rendered as escaped `(…)` `IN`-lists). A non-scalar value with no SQL form (a JSON object, or an empty array) is rejected rather than emitted raw. @@ -210,7 +223,7 @@ The package's design invariants — stdout always 100%, WARN+ERROR always export ### `query/` — Structured Query Engine - **ast.go** — `StructuredQuery` AST types: columns, aggregations, filters, group by, order by, limit, time range. -- **builder.go** — `Build()` converts AST to parameterized SQL. It is the single chokepoint that validates every referenced identifier against the schema **and** authorizes every column reference — projection, aggregation args, filters, group_by, order_by, time_range — against the role's column allowlist (the [#223](https://github.com/Wave-RF/WaveHouse/issues/223) hard cap). A full-row read is requested with `select_all`, which expands to the role's allowed columns rather than emitting a raw `SELECT *`; an omitted projection selects nothing, and `*` in `columns` is a literal column name. Every identifier is backtick-quoted via `internal/chsql` (`QuoteIdent`) so any ClickHouse-legal name is accepted — a name containing `?` is refused fail-closed ([#279](https://github.com/Wave-RF/WaveHouse/issues/279)). The role's row-level-security predicate and `max_rows` cap are emitted by `Build()` itself, as part of the WHERE and LIMIT assembly — policy SQL is never spliced into rendered text ([#322](https://github.com/Wave-RF/WaveHouse/issues/322)). Timestamp bucketing for cache optimization. +- **builder.go** — `Build()` converts AST to parameterized SQL. It is the single chokepoint that validates every referenced identifier against the schema **and** authorizes every column reference — projection, aggregation args, filters, group_by, order_by, time_range — against the role's column allowlist (the [#223](https://github.com/Wave-RF/WaveHouse/issues/223) hard cap). A full-row read is requested with `select_all`, which expands to the role's allowed columns rather than emitting a raw `SELECT *`; an omitted projection selects nothing, and `*` in `columns` is a literal column name. Every identifier is backtick-quoted via `internal/chsql` (`QuoteIdent`) so any ClickHouse-legal name is accepted — a name containing `?` is refused fail-closed, because `Build` emits positional placeholders that a later pass rewrites to ClickHouse's named parameters ([#279](https://github.com/Wave-RF/WaveHouse/issues/279)). The role's row-level-security predicate and `max_rows` cap are emitted by `Build()` itself, as part of the WHERE and LIMIT assembly — policy SQL is never spliced into rendered text ([#322](https://github.com/Wave-RF/WaveHouse/issues/322)). Timestamp bucketing for cache optimization. ### `settings/` — Settings Directory @@ -235,7 +248,7 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi ### `chsql/` — ClickHouse SQL Helpers -- **chsql.go** — Dependency-free ClickHouse SQL helpers shared by `query/` and `policy/`, kept in their own package to break an import cycle. `QuoteIdent` is the single place every identifier — column, table, alias — becomes SQL text: always backtick-quoted and escaped, so any ClickHouse-legal name (dots, spaces, unicode, keywords) is safe. `BindUnsafe` reports whether a name contains a literal `?`, which would desync clickhouse-go's positional binder; such names are rejected fail-closed rather than silently mis-bound. +- **chsql.go** — Dependency-free ClickHouse SQL helpers shared by `query/` and `policy/`, kept in their own package to break an import cycle. `QuoteIdent` is the single place every identifier — column, table, alias — becomes SQL text: always backtick-quoted and escaped, so any ClickHouse-legal name (dots, spaces, unicode, keywords) is safe. `BindUnsafe` reports whether a name contains a literal `?`, which would desync the positional-to-named parameter rewrite; such names are rejected fail-closed rather than silently mis-bound. `IntegerType` and `StrictInt` pick and render the strict round-trip cast a policy claim is compared through on an integer column, identically for the query builder and the type layer, so a claim that does not fit the column's type is NULL and matches nothing instead of wrapping. ### `keyenc/` — Key Escaping @@ -259,25 +272,35 @@ Client POST /v1/ingest?table={table} unsupported, or if declarations disagree; before the body is read) → Read the whole body into a pooled buffer, bounded by the 16 MiB cap (413 before any record is processed, so nothing is published) - → Validate JSON body against schema (type checks, required columns) - → Policy column rules + check clauses (disallowed columns rejected; - claim-derived values enforced or injected) - → Canonicalize top-level DateTime/DateTime64 column values to RFC 3339 UTC - (rewrites the payload so every consumer shares one spelling; fail-open — - an unparseable value passes through verbatim for ClickHouse's parser to judge) + → Compile the ROLE's schema for the request's tenant: the columns it may + insert, plus a DEFAULT per _eq check column carrying the claim (cached per + generation+shape). A tenant the engine cannot answer for (no artifact for + its ClickHouse line, or a server time zone that differs from the one this + process opened that line with) → 503 + Retry-After: 5, before the body is read + → Validate the whole body through chtypes (internal/typelayer), one call per + request: ClickHouse's own parser type-checks, coerces, and fills DEFAULTs + (including now()) per record. A rejected record carries ClickHouse's real + error code (exception_code) and message — an unknown field, a + MATERIALIZED/ALIAS/EPHEMERAL column, or a column this role may not write is + 117; a record chtypes cannot answer for is a distinct "declined" outcome + (422), never a data rejection + → Evaluate the role's check clauses over the accepted rows with one compiled + chtypes filter — false is 403 for that record, unevaluable is 422 → Optional dedupe: resolve the id (configurable ID field; a row missing it or setting it to null is published un-deduped + logged/counted, or rejected - under require_id) - → Encode the record; the steps below run per window of up to 256 records + under require_id) — deliberately after validation, so a chtypes-rejected + record is never marked seen + → The steps below run per window of up to 256 accepted records → Reserve the window's (tenant, table, id) keys in one call: a duplicate is skipped, an id another request holds → 503 + Retry-After (dedupe.lease, 30s by default), a store that cannot answer → 503 + Retry-After: 5 - → Publish each record to NATS JetStream (ingest.{tenant}.{table}), a deduped - one under its idempotency key + → Publish each ClickHouse-rendered row (DEFAULTs filled, MATERIALIZED/ + ALIAS/EPHEMERAL columns absent) to NATS JetStream (ingest.{tenant}.{table}), + a deduped one under its idempotency key → Commit the published ids in one call; on a failed publish, commit the records before it and release the rest (a publish whose outcome is unknown keeps its claim until the lease lapses) - → 200 OK returned immediately + → 200 OK returned immediately, per-record outcomes in the response body → (If the tenant's NATS stream is full, or not open: 503 + Retry-After header, the id released) @@ -291,8 +314,10 @@ Ingest worker pipeline (StartIngestWorker): → Batch events per tenant table → At flush, a batch whose tenant has no ClickHouse connection is not inserted (parkBatch takes it to the DLQ switch whole); otherwise bulk INSERT to ClickHouse - (INSERTs pin date_time_input_format=best_effort — the server default since - ClickHouse 26.5; see /ingest-pipeline for the basic-vs-best_effort divergence) + (INSERTs pin the same parsing settings chtypes compiled the row with — + date_time_input_format=best_effort, input_format_null_as_default=1 — plus + async_insert=0, so per-row error attribution isn't lost to an async flush + wait; see /ingest-pipeline for detail) → On success: DoubleAck messages → On failure ClickHouse could not take (down, overloaded, read-only, denied — chconn.Classify): NakWithDelay the batch under the pool's backoff (the table's, for a @@ -362,7 +387,7 @@ Client POST /v1/ops/query (browser, CDN, corp proxy) caches the result. ``` -The proxy-pattern wins are: zero classification logic on the WaveHouse side (no `IsMutation` heuristic to maintain), and any ClickHouse statement type — including verbs added in future versions and inline FORMAT overrides — works without WaveHouse code changes. Multi-statement input (`SELECT 1; TRUNCATE t`) is supported when the upstream ClickHouse has multi-query enabled, which is the default on recent versions; older or restrictively-configured servers will return a clear error from ClickHouse itself for the second statement. The proxy buffers the response in memory with a 64 MiB cap (502 with `clickhouse response exceeded N bytes` on overflow, to keep a runaway `SELECT *` from pinning RAM on the API server), and passes ClickHouse's `Content-Type` through when an inline `FORMAT` directive overrides the default JSON envelope. The structured query endpoint and pipes still go through `clickhouse-go`'s native driver (Query/Exec) for performance and to keep the cached row-array shape consistent. +The proxy-pattern wins are: zero classification logic on the WaveHouse side (no `IsMutation` heuristic to maintain), and any ClickHouse statement type — including verbs added in future versions and inline FORMAT overrides — works without WaveHouse code changes. Multi-statement input (`SELECT 1; TRUNCATE t`) is supported when the upstream ClickHouse has multi-query enabled, which is the default on recent versions; older or restrictively-configured servers will return a clear error from ClickHouse itself for the second statement. The proxy buffers the response in memory with a 64 MiB cap (502 with `clickhouse response exceeded N bytes` on overflow, to keep a runaway `SELECT *` from pinning RAM on the API server), and passes ClickHouse's `Content-Type` through when an inline `FORMAT` directive overrides the default JSON envelope. The structured query endpoint and pipes reach ClickHouse the same way, over its HTTP interface, but ask for `default_format=JSONEachRow` and bind their values as named `{pN:String}` parameters. ClickHouse renders the JSON; WaveHouse frames the lines into an array and caches those bytes, so a value is spelled by the connected server with no WaveHouse conversion table in between. The native driver now serves only schema discovery and the readiness ping. ### Streaming Path @@ -392,7 +417,8 @@ Client GET /v1/stream → fan the finished frame to every Subscriber of that (topic, role); a role carrying a row-level filter delivers per subscriber instead: the shared frame goes only to subscribers whose JWT claims admit - the row (RowVisible) + the row (internal/typelayer's Row.Visible, compiled and evaluated by + the same engine as the server's WHERE clause) → a connection whose column list drifts (a schema change mid-stream) is re-announced before the next row, per connection → Live and replay track that drift in SEPARATE state and do not reconcile @@ -404,7 +430,9 @@ Client GET /v1/stream under the wrong names until it reconnects → Handler drains keepalives + event frames from one byte-pump → client → Policy filtering (historical + live): denied tables skipped, denied - columns stripped, row filter evaluated per subscriber against claims. + columns stripped, row filter compiled and evaluated per subscriber + against claims by internal/typelayer (fail closed on a compile/parse + error, a missing policy column, schema drift, or an unavailable engine). Column projection runs once per role (Hub.Broadcast) — per-subscriber work only where a row filter makes visibility per-connection; replay shares the same column policy + row check but projects per-connection @@ -414,7 +442,7 @@ Client GET /v1/stream | Component | Technology | Purpose | | --------- | ---------- | ------- | -| Language | Go 1.26 | Core runtime | +| Language | Go 1.27 | Core runtime | | HTTP Router | Chi v5 | Request routing and middleware | | Authentication | golang-jwt v5 + keyfunc v3 | JWT (HMAC + JWKS) parsing and validation | | Analytics DB | ClickHouse | Primary data store + schema source of truth | @@ -423,5 +451,6 @@ Client GET /v1/stream | Shared Cache | [rueidis](https://github.com/redis/rueidis) | Redis-compatible client for the shared backend (`cache.backend: redis`) | | Embedded KV | Pebble | Optional deduplication | | Config | cleanenv | YAML + env var config loading | -| Release | GoReleaser | Cross-platform binary builds | -| Containers | Docker (distroless) | Minimal production images | +| Type engine | [chtypes](https://github.com/wave-rf/chtypes) (cgo/dlopen) | In-process ClickHouse parser: ingest validation/coercion + row-level-security compilation | +| Release | GoReleaser | Binary builds for Linux amd64/arm64 and macOS arm64 | +| Containers | Docker (`distroless/cc`, glibc) | Minimal production images | diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index 69dfdcc2..27740d2f 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -128,7 +128,7 @@ Boot logs each of these at `WARN` rather than refusing. The first two are right ### Process roles -By default one process does all the work. `roles` splits it, so that the API and the background workers can run in separate processes, for example one Kubernetes Deployment per role (see [Deployment](/deployment#one-deployment-per-role)). The binary and its entry point are the same for every role; only this key differs. +By default one process does all the work. `roles` splits it, so that the API and the background workers can run in separate processes, for example one Kubernetes Deployment per role (see [Deployment](/deployment#one-deployment-per-role)). The binary and its entry point are the same for every role; only this key differs. Only a process running the `api` role loads the [chtypes artifact](/deployment#chtypes-artifacts); an ingest-only or sweeper-only process needs none installed. | YAML Key | Env Var | Default | Description | | --- | --- | ------- | ----------- | @@ -168,6 +168,7 @@ Only the secret and the connection ceiling are boot config. The wiring — nativ | --- | --- | ------- | ----------- | | `clickhouse.password` | `WH_CH_PASSWORD` | *(empty)* | Authentication password, combined with the settings directory's `clickhouse.username` on every (re)connect. A secret, so it never lives in a tracked JSON file; rotating it is a restart. | | `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the `max_open_conns` of the open pools — one per distinct connection tuple among the served tenants, see [the settings directory](/settings-directory#clickhouse) — must not add up to more. Pools above it at boot refuse to start, naming the sum and the ceiling; on a reload a pool resized above it is refused and keeps its size, and a new pool that would cross it is not opened — its tenants keep the pool they had, or have none — both logged (the reload itself still reports `adopted`) and retried by the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. | +| `clickhouse.chtypes_registry` | `WH_CHTYPES_REGISTRY` | *(empty)* | Directory holding the chtypes artifacts (one `/` per ClickHouse line). Empty defers to the chtypes search path — `$CHTYPES_REGISTRY`, `~/.cache/chtypes/artifacts/abi6/-`, then the system directories. Either way a library is opened lazily, on first use of its line; an explicit directory is searched first, then the rest of the path. Boot-tier: changing it is a restart. See [chtypes artifacts](/deployment#chtypes-artifacts). | ### Server-side resource limits diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 8bcc248d..b1e02c80 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -7,11 +7,11 @@ sidebar: order: 10 --- -How to run WaveHouse in production — single binary, Docker images, releases, health checks, and the required ClickHouse schema. +How to run WaveHouse in production — one binary, Docker images, releases, health checks, and the required ClickHouse schema. -## Single binary +## One binary, plus a per-version artifact -WaveHouse runs as one process with embedded NATS and optional Pebble dedup. The only external dependency is ClickHouse, unless you select a shared backend: [`mq.backend: nats`](#external-nats) (and `coord.backend: nats`), [`cache.backend: redis`](#multiple-instances-and-the-shared-cache) or [`dedupe.backend: dynamodb`](#a-shared-dedupe-table-on-dynamodb). +WaveHouse runs as one process with embedded NATS and optional Pebble dedup. At start an API-role process also loads a second artifact — a per-ClickHouse-version shared library that runs ClickHouse's own parser in-process for ingest validation and row-level security (via [chtypes](#chtypes-artifacts)); a process that runs only the ingest worker or the sweeper loads none. The only external network dependency is ClickHouse, unless you select a shared backend: [`mq.backend: nats`](#external-nats) (and `coord.backend: nats`), [`cache.backend: redis`](#multiple-instances-and-the-shared-cache) or [`dedupe.backend: dynamodb`](#a-shared-dedupe-table-on-dynamodb). ### Quick Start with Docker Compose @@ -78,7 +78,7 @@ docker build -f deployments/Dockerfile -t wavehouse:latest . This builds the runtime image `wavehouse:latest`. (The published `ghcr.io` images are built by GoReleaser from `deployments/Dockerfile.goreleaser`, not this command — see Registry below.) -All images use multi-stage builds (Go Alpine builder → distroless runtime) for minimal attack surface. +All images use multi-stage builds (`golang:1.27-bookworm` glibc builder → `gcr.io/distroless/cc-debian12` runtime — cgo needs glibc, so the previous Alpine/musl builder and `distroless/static` runtime no longer work) for minimal attack surface. The build also fetches the [chtypes artifact(s)](#chtypes-artifacts) pinned in `chtypes.lock` into the image. ### Registry @@ -104,18 +104,52 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml ``` +## chtypes artifacts + +**What it is.** WaveHouse validates ingest data, coerces types, substitutes `DEFAULT`s, and evaluates row-level security by running ClickHouse's own parser in-process, through [chtypes](https://github.com/wave-rf/chtypes) (`github.com/wave-rf/chtypes/go` v0.4.0). The parser itself ships as a shared library (`libchtypes.so` / `.dylib`) built per **ClickHouse minor line** (e.g. `26.6`) and per platform. There is no nearest-version fallback: each tenant's table set is compiled from its own schema refresh against the artifact matching that tenant's ClickHouse server, and a library is opened lazily, the first time a tenant on that line is bound. + +**Which processes load it.** Only processes with the `api` role — the ones that serve ingest and the stream. A process that runs only the ingest worker or the sweeper loads no artifact and boots without one installed. An API process refuses to start when no artifact is installed at all. + +**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with a message naming both zones; run such tenants in a separate process, or align the servers' `timezone` setting. See [API → Ingest error responses](/api#error-responses) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). + +**Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse does not autofetch on a miss in production — an unmatched line is a boot-time or refresh-time failure, not a background download. + +**Size.** Each artifact is roughly 160–290 MB on disk; a running process holding several loaded versions (e.g. across a rolling ClickHouse upgrade) costs roughly 120 MB of resident memory per loaded version (the chtypes multi-version guide's figure; a library is opened on first use of its line, not at registry construction). + +**Docker images** ship the artifact(s) baked in: the image build fetches whatever `chtypes.lock` names (see below), so a container never needs network access to ClickHouse's artifact store at runtime. `WH_CHTYPES_REGISTRY` (default `/opt/chtypes/artifacts`) points at the directory inside the image. + +**Release archives and `go install` / building from source** do not carry or fetch an artifact — only the Docker images bake one in. See the [README's `go install` caveat](https://github.com/Wave-RF/WaveHouse#c-go-install-binary-no-docker). Fetch one yourself before first run: + +```bash +scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch --frozen --lock chtypes.lock 26.6 +``` + +or, for a line not in the repo's lock file: + +```bash +go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch +``` + +### Pinning with `chtypes.lock` + +`chtypes.lock`, checked in at the repo root, records the exact artifact file and SHA-256 per platform/line the project builds and tests against. CI restores from it with `--frozen` (refusing anything the lock doesn't name) rather than fetching the rolling artifact release, so a pipeline never silently starts testing a new build. Refresh it deliberately — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch --lock chtypes.lock --platform `, once per platform (`darwin-arm64`, `linux-amd64`, `linux-arm64`), without `--frozen` — and commit the result; don't regenerate it implicitly. + +A lock is specific to the SDK's ABI revision (6 at v0.4.0): the fetcher never selects a build from another revision, so after an SDK bump that changes the revision, `--frozen` fails (`CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED`) until the lock is regenerated the same way, and the CI cache key and path (`abi6`) move with it. + ## Releases Releases are built with [GoReleaser](https://goreleaser.com/). The configuration is in `.goreleaser.yaml`. The release archives attached to each GitHub Release carry a signed [Sigstore](https://www.sigstore.dev/) build-provenance attestation — verify a downloaded archive with `gh attestation verify --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml`. (This covers the prebuilt archives, not `go install`, which compiles from source.) ### Supported Platforms +The binary requires cgo (dlopen only — no static link to the chtypes artifact) and glibc 2.34 or later, which sets the supported platform matrix: + | OS | Architecture | | -- | ----------- | | Linux | amd64, arm64 | -| macOS | amd64, arm64 | -| Windows | amd64, arm64 | -| FreeBSD | amd64, arm64 | +| macOS | arm64 only | + +Windows, FreeBSD, and darwin/amd64 are no longer built — there is no chtypes artifact for them, and the binary cannot run without one. If you need one of these, [open an issue](https://github.com/Wave-RF/WaveHouse/issues) describing your use case. ### Creating a Release @@ -607,7 +641,7 @@ For local development, `docker compose -f deployments/compose/dependencies.yaml WaveHouse uses a **Bring Your Own Schema** model. You create your tables in ClickHouse with whatever columns and engines you need. WaveHouse discovers the schemas automatically via `system.columns` and validates ingest data against them — see [Schema Validation](/api#post-v1ingesttabletable--ingest-data) for the rules a record must satisfy. -Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED` or `ALIAS` column is computed by ClickHouse and cannot be inserted: omit it from your records, and a record that names one is rejected. An `EPHEMERAL` column is the awkward one — it *is* insertable, but it is never stored and no query can read it back, so it is only useful as an input to another column's `DEFAULT` expression, and a policy `check` naming one is refused outright. And a `Nullable(T) DEFAULT …` column never takes its default through ingest: an omitted key stores `NULL`, not the default — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for why. A **non-nullable** column with a default is unaffected. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). +Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names one is rejected with ClickHouse's own code (117) rather than published; a policy `check` naming one is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's implicit zero value where none is declared, evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). Example table: @@ -720,7 +754,7 @@ The streaming surface loses something too, more quietly. SSE gap-fill (`?since=` Three audits belong **before** the drain, because none of them announces itself afterwards: -- **`Nullable(T) DEFAULT …` columns now store `NULL` where they took their default.** A positional row has one slot per insertable column and no way to say *absent*, so a key the record omits rides as an explicit `null`. `input_format_null_as_default=1` turns that back into the default for a **non-nullable** column, but ClickHouse stores `NULL` on a nullable one whatever the setting says — only an absent key ever took the default. Following this runbook exactly still changes what lands in those columns, silently. See [the ingest note](/ingest-pipeline#the-journey-of-one-event). +- **The wire envelope's `row` is positional** (`columns` names each slot), so a message from before this migration and one from after it look the same shape-wise; what changed underneath is how an omitted field is resolved into that slot — see [the ingest note](/ingest-pipeline#the-journey-of-one-event) for the current behavior. - **Policy `check` blocks are now validated against the table.** A `check` naming a column the table lacks, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is a per-record `403` on *every* insert by that role. `wavehouse validate` cannot catch it — it never sees the ClickHouse schema — so audit them against their tables first. See [Access control → Insert checks](/access-control#insert-checks). - **Every `WH_*` variable the binary does not bind refuses boot.** The old binary ignored a variable it did not read; the new one names every unbound one and exits before it opens the queue, so a pod spec or compose file that still carries one comes back from the upgrade as a container that will not start. Diff the environment against the [Configuration Reference](/configuration) first: a `WH_*` variable that is not in its tables is unbound, and whatever it used to configure now lives in the [settings directory](/settings-directory) or is gone. A Kubernetes Service in the pod's namespace named `wh` or `wh-*` counts too: it injects link variables under the `WH_` prefix (`WH_SERVICE_HOST` and `WH_PORT` for `wh`, `WH_FOO_SERVICE_HOST` and `WH_FOO_PORT` for `wh-foo`), so set `enableServiceLinks: false` on the pod spec. diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index 4738b3dd..a33b205d 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -13,7 +13,7 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: | Tool | Required version | Why | Install | | ---- | ---------------- | --- | ------- | -| **Go** | 1.26+ (matches `go.mod`) | Compiles `cmd/wavehouse`; also runs the pinned `tool` deps (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `deadcode`, `gsa`, `goda`) via `go tool` | [go.dev/dl](https://go.dev/dl/) | +| **Go** | 1.27+ (matches `go.mod`) | Compiles `cmd/wavehouse` with cgo enabled (needed by chtypes' dlopen shim — a C toolchain and glibc must be present); also runs the pinned `tool` deps (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `deadcode`, `gsa`, `goda`) via `go tool` | [go.dev/dl](https://go.dev/dl/) | | **GNU Make** | **4.0+** | The Makefile uses `--output-sync=target` (Make 4 only) and bash-pinned recipes. macOS ships with BSD Make 3.81, which **will not work** | macOS: `brew install make` then use `gmake` or put `$(brew --prefix make)/libexec/gnubin` on your PATH. Linux: usually already installed | | **bash** | 4+ recommended | Recipes are pinned to `bash`; the helper scripts under `scripts/` use `set -euo pipefail` and bash arrays | macOS default is bash 3.2 (works for current recipes, but `brew install bash` is safer); Linux distros ship 4+ | | **Docker** *(or Podman)* | Engine 20.10+ with the Compose **v2** plugin (`docker compose`, no hyphen) | Compose stacks under `deployments/compose/`; the E2E and integration suites boot ClickHouse and a Redis via testcontainers (no compose file), the integration suite also dynamodb-local, and the integration suite also runs the shared cache backend against Redis, Valkey, Dragonfly (pulled from `docker.dragonflydb.io`) and a one-node Redis Cluster | [Docker Desktop](https://docs.docker.com/get-docker/), [colima](https://github.com/abiosoft/colima), or [Podman](https://podman.io) with `podman-compose` / the `podman compose` plugin. The testcontainers Go library also honors `DOCKER_HOST` for rootless Podman setups | @@ -21,6 +21,16 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: | **pnpm** | 11.21+ (pinned via `packageManager` in the root `package.json`) | Package manager for the TypeScript SDK, E2E test harness, and docs site (managed as a single pnpm workspace from the repo root); `make build-ts`, `make test-ts`, `make test-e2e`, `make build-docs`, `make dev-docs`, `make preview-docs` all shell out to `pnpm` | `corepack enable && corepack prepare pnpm@11.21.0 --activate` (recommended), or `npm i -g pnpm` | | **git** + **curl** | any recent | `git` for source + version metadata in builds; `curl` is used by the Makefile to fetch the pinned `golangci-lint` binary into `.bin/` | usually preinstalled | +### The chtypes artifact — fetch it once per machine + +`internal/typelayer` loads a per-ClickHouse-version shared library at start to run ingest validation and row-level security through ClickHouse's own parser (see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)). It is not source code and `make tools` does not fetch it for you — pull it once with: + +```bash +scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch --frozen --lock chtypes.lock 26.6 +``` + +It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–290 MB — expect the first run to take a minute or two. Without it, `make dev` / `make test` / `make test-e2e` fail closed (a `503` on ingest, every stream row withheld) until a matching artifact exists for the ClickHouse line the tests or your local server run against. + ### Auto-installed by `make tools` Run `make tools` once after cloning to populate everything that doesn't have to be on your PATH: @@ -34,7 +44,7 @@ Run `make tools` once after cloning to populate everything that doesn't have to ### Verify your setup ```bash -go version # go1.26+ +go version # go1.27+ make --version # GNU Make 4.x docker compose version node --version # v22.x (matches .nvmrc and CI) @@ -547,10 +557,9 @@ Run `make help` to see all targets. Key ones: | `make release-sdk-go VERSION=X.Y.Z` | Tag a Go SDK release — `go get` (pending [#434](https://github.com/Wave-RF/WaveHouse/pull/434)) | | **Analysis** (informational, not in CI) | | | `make size` | Binary size analysis → `tmp/analysis/` (text + SVG + interactive HTML) | -| `make audit-cgo` | Audit dependency tree for C files (builds use `CGO_ENABLED=0`) | | `make deadcode` | Find unreachable functions | | `make dep-cut` | Top cuttable deps by transitive weight (`LIMIT=N` to override) | -| `make binary-analysis` | Combined: `size` + `audit-cgo` + `deadcode` | +| `make binary-analysis` | Combined: `size` + `deadcode` | | **Cleanup** (tiered — compose explicitly for partial resets) | | | `make clean` | Build outputs only (`bin/`, `dist/`, `clients/ts/dist/`, `docs/dist/`, `docs/.dev-dist/`) | | `make clean-test` | Test outputs only (`tmp/` — coverage data, logs, NATS state) | @@ -619,7 +628,7 @@ The **tag is the version**, everywhere: | Component | Where the version comes from | | --- | --- | -| Server | GoReleaser's `-ldflags` at build time, from the tag | +| Server | GoReleaser's `-ldflags` at build time, from the tag (`goreleaser build --single-target`, once per platform) | | Go SDK | The tag itself — a Go module has no version file | | TypeScript SDK | `publish-npm.yml` stamps `clients/ts/package.json` from the tag before publishing | @@ -639,7 +648,7 @@ Tag globs are anchored at the start of the ref name, so `v*` never matches a `cl ### What a release publishes -- **Server —** a **GitHub Release** with the cross-compiled archives (linux/darwin/windows/freebsd × amd64/arm64; `.zip` on Windows, `.tar.gz` elsewhere) and `checksums.txt`. A tag carrying a prerelease suffix (`v0.1.0-alpha.1`) is marked as a GitHub pre-release, so it never takes the "Latest release" badge from a shipped stable version. +- **Server —** a **GitHub Release** with archives for the three supported platforms (linux/amd64, linux/arm64, darwin/arm64 — cgo's dlopen requirement and the lack of a chtypes artifact elsewhere dropped Windows, FreeBSD, and darwin/amd64; see [Deployment → Supported Platforms](/deployment#supported-platforms)), each `.tar.gz`, and `checksums.txt`. A tag carrying a prerelease suffix (`v0.1.0-alpha.1`) is marked as a GitHub pre-release, so it never takes the "Latest release" badge from a shipped stable version. - **Both —** **release notes generated by GitHub** from the PRs merged since the previous tag *in the same family* — one line per PR, since `main` is squash-merged, grouped into the categories defined in [`.github/release.yml`](https://github.com/Wave-RF/WaveHouse/blob/main/.github/release.yml). Grouping is by **PR label**: `github_actions` / `documentation` are applied automatically by `actions/labeler`, but `breaking-change`, `security`, `bug`, and `enhancement` are applied by hand — an unlabelled PR lands in "Other changes". Dependabot is split out by **author** rather than by label, because the labels `actions/labeler` applies by path — `github_actions`, `documentation` — mark our own PRs too; our CI work gets its own "CI & build" section — ordered above Documentation, since a CI PR here nearly always updates docs too — and Dependencies is pure Dependabot residue. **Any category keyed on a label a Dependabot PR can carry needs that author exclude** — labeler's path labels *and* the ecosystem labels Dependabot applies itself (`dependencies`, `javascript`, `go`, `github_actions`; `javascript` is in neither `labeler.yml` nor our categories) — or that category intercepts bumps before they reach the `📦 Dependencies` catch-all. `CHANGELOG.md` is *not* the source of the release body; it is the longer-form record of why each change was made. - **Server —** a **GHCR image** at `ghcr.io/wave-rf/wavehouse`, with two tags: the immutable `:vX.Y.Z`, and one moving *channel* pointer. A stable release moves `:latest`; a prerelease moves `:alpha` / `:beta` / `:rc` / `:next` instead, matching the npm dist-tag it would get. The channel comes from the **first** prerelease identifier, matched **exactly**: `v0.2.0-rc.1` → `:rc`, while `-alpha1`, `-preview.1`, or any other form → `:next`. `scripts/ci/release-channel.sh` is the single rule every publisher uses, so `ghcr.io/wave-rf/wavehouse:rc` and `@wavehouse/sdk@rc` can't drift apart. **A prerelease-only project therefore has no `:latest` tag** — that is deliberate; `:latest` starts existing when the first stable release ships. - **TypeScript SDK —** an **npm publish** of `@wavehouse/sdk` under `latest` (stable) or `alpha`/`beta`/`rc`/`next` (prerelease), plus its own GitHub Release. @@ -663,14 +672,32 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:v0.1.0 \ `--signer-workflow` is not optional garnish: `--repo` alone accepts an attestation produced by *any* workflow in the repo. Same point, and the `:dev` equivalent, in [Deployment → Registry](/deployment#registry) and `SECURITY.md`. :::note[Releasing from the GitHub UI instead] -Publishing a release from **Releases → Draft a new release** creates the tag, which fires the same workflow — so it works, and GoReleaser's default `mode: keep-existing` (the key is not set in `.goreleaser.yaml`) means it will not overwrite notes you wrote. Two things the `make` targets do for you and the UI does not: none of the preflight checks run, and you must click **Generate release notes** yourself, because a body you publish empty stays empty. +Publishing a release from **Releases → Draft a new release** creates the tag, which fires the same workflow — so it works, and it will not overwrite notes you wrote: `release.yml` checks `gh release view` first and, when the release already exists, only uploads the assets. (That check also makes the job re-runnable, which is why it is not conditional on how the release was created.) Two things the `make` targets do for you and the UI does not: none of the preflight checks run, and you must click **Generate release notes** yourself, because a body you publish empty stays empty. ::: +### How the server release is built + +Since the switch to cgo the three binaries are built on **three native runners**, not cross-compiled from one: + +| Target | Runner | +| --- | --- | +| `linux/amd64` | `ubuntu-latest` | +| `linux/arm64` | `ubuntu-24.04-arm` | +| `darwin/arm64` | `macos-latest` | + +All three are free for public repositories. Each runs `goreleaser build --single-target --output dist/wavehouse` — so `.goreleaser.yaml` is still the one place the build's `-ldflags`, binary name and supported platform set are declared — and uploads the binary as a run artifact. A final `ubuntu-latest` job assembles everything: the three `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image via `docker buildx build` over [`deployments/Dockerfile.goreleaser`](https://github.com/Wave-RF/WaveHouse/blob/main/deployments/Dockerfile.goreleaser), the GitHub Release, and the provenance attestations. + +**Why not one runner and cross-compilers?** cgo needs a C toolchain per target, and for darwin that means real Apple SDK headers. `zig cc -target aarch64-macos` cross-compiles most Go programs happily, but not this one: `prometheus/client_golang`'s darwin process collector is a C file that `#include`s ``, which zig does not ship and which cannot legally be fetched onto a GitHub-hosted Linux runner. Measured, with and without `-tags netgo,osusergo`; the two GoReleaser features that would solve it — split/merge and `builder: prebuilt` — are Pro-only, and OSS `goreleaser release` has no `--skip=build`, so there is no way to have GoReleaser assemble a release from binaries built elsewhere. The two Linux targets *can* be cross-compiled (`gcc-aarch64-linux-gnu` works), but building them natively alongside darwin costs nothing extra and keeps one rule instead of two. + +A consequence worth knowing when you file a bug: the released Linux binaries are **dynamically linked against glibc**, minimum `GLIBC_2.34` (measured on `ubuntu-24.04`, both architectures) — Debian 12, Ubuntu 22.04 and RHEL 9 or newer. The pre-cgo builds were static and ran anywhere. The container images are unaffected; their `distroless/cc-debian12` base is glibc 2.36. + +`goreleaser-validate.yml` is the PR-time proof of all of this. On a change to `.goreleaser.yaml`, `go.mod`/`go.sum`, `chtypes.lock`, `scripts/fetch-chtypes.sh`, `deployments/Dockerfile.goreleaser` or the release workflows it runs `goreleaser check`, the same three-runner matrix in `--snapshot` mode, and a real multi-arch image build (to `--output type=cacheonly`, so nothing is pushed). It is advisory, not a required check. + ## The `dev` channel Between releases, every push to `main` republishes both artifacts so `@dev` always means "current `main`": -- **`ghcr.io/wave-rf/wavehouse:dev`** — a rolling pointer, plus an immutable `:dev-` (pruned after 30 days by `cleanup-ghcr.yml`, newest 5 always kept). Built by the same GoReleaser pipeline with `WAVEHOUSE_DEV=1`, which suppresses the GitHub Release. Note a Docker tag is only a pointer: `docker run …:dev` reuses a stale local image unless you `docker pull` first or pass `--pull=always`. +- **`ghcr.io/wave-rf/wavehouse:dev`** — a rolling pointer, plus an immutable `:dev-` (pruned after 30 days by `cleanup-ghcr.yml`, newest 5 always kept). Built by `publish-dev.yml`, the same shape as a real release minus everything that isn't the image: two native Linux build jobs and one push job, no archives and no GitHub Release. Note a Docker tag is only a pointer: `docker run …:dev` reuses a stale local image unless you `docker pull` first or pass `--pull=always`. - **`@wavehouse/sdk@dev`** — `0.0.1-dev..h`, published only when the published package actually changes. The trailing hash covers every file `npm pack` would ship — the built `dist/`, `package.json` minus its `version`, and the bundled `README`/`LICENSE` — so a push whose package would be byte-identical to the current `dev` publish is skipped, while a change to `exports`, `files`, `bin`, or `engines` republishes even though `dist/` is untouched. The `` is load-bearing, not decoration. `npm install …@dev` records a *range* in your `package.json`, not the dist-tag, so what you get on the next install is the highest version matching that range. Under the old `0.0.0-dev.h` scheme, semver's lexical ordering of alphanumeric prerelease identifiers meant the newest publish routinely wasn't the highest one, and a range could resolve *backwards* — an `@dev` install landed on a two-month-old build ([#475](https://github.com/Wave-RF/WaveHouse/issues/475)). A numeric identifier compares numerically, so the channel now orders by publish time. The `0.0.1` base keeps the channel below every real release (so a dev build can never satisfy `^0.1.0`) and above the legacy `0.0.0-dev.*` publishes, which npm's 72-hour unpublish window makes permanent. diff --git a/docs/src/content/docs/getting-started.md b/docs/src/content/docs/getting-started.md index 667137d1..d2b91168 100644 --- a/docs/src/content/docs/getting-started.md +++ b/docs/src/content/docs/getting-started.md @@ -5,13 +5,13 @@ sidebar: order: 2 --- -Run WaveHouse locally in under five minutes. WaveHouse ships as a single binary with ClickHouse as the only external dependency; this walkthrough covers ingest, query, and real-time streaming. +Run WaveHouse locally in under five minutes. WaveHouse ships as one binary plus the per-ClickHouse-version [chtypes artifact](/deployment#chtypes-artifacts) it loads at start, with ClickHouse as the only external network dependency; this walkthrough covers ingest, query, and real-time streaming. ## Prerequisites - **Docker** — for running ClickHouse (and optionally WaveHouse itself). - **curl** and **jq** (optional) — for poking the API. -- **Go 1.26+** — only required if you want to build from source; skip it for the Docker path below. +- **Go 1.27+** — only required if you want to build from source; skip it for the Docker path below. Building from source also requires cgo (a C toolchain) and glibc — see [Deployment → Supported Platforms](/deployment#supported-platforms). ## 1. Start WaveHouse @@ -62,7 +62,7 @@ curl -s -X POST "http://localhost:8080/v1/ingest?table=clicks" \ # → {"ok":true} ``` -WaveHouse validates the body against the ClickHouse schema before acknowledging. Unknown fields, type mismatches, and missing required columns are rejected with a `400`. +WaveHouse validates the body against the ClickHouse schema before acknowledging — using ClickHouse's own parser, running in-process (`internal/typelayer`, via [chtypes](/deployment#chtypes-artifacts)), so a rejection carries ClickHouse's own error code and message, the same as a native `INSERT` would produce. Unknown fields and type mismatches are rejected with a `400`. ## 4. Query diff --git a/docs/src/content/docs/index.mdx b/docs/src/content/docs/index.mdx index 205a6f84..1033bfaa 100644 --- a/docs/src/content/docs/index.mdx +++ b/docs/src/content/docs/index.mdx @@ -1,6 +1,6 @@ --- title: Real-time API gateway for ClickHouse -description: The open-source real-time API gateway for ClickHouse — schema-aware ingest, async batching, real-time streaming, and tiered query caching in a single binary. +description: The open-source real-time API gateway for ClickHouse — schema-aware ingest, async batching, real-time streaming, and tiered query caching in one binary. template: splash # Homepage-only structured data: lets Google/GitHub attach the canonical # SoftwareSourceCode + Organization entities to the project's root URL. @@ -10,10 +10,10 @@ head: attrs: type: application/ld+json content: | - {"@context":"https://schema.org","@graph":[{"@type":"Organization","@id":"https://wave-rf.com/#org","name":"Wave RF","url":"https://wave-rf.com","email":"hello@wave-rf.com","sameAs":["https://github.com/Wave-RF"]},{"@type":"SoftwareSourceCode","@id":"https://wavehouse.dev/#project","name":"WaveHouse","description":"The open-source real-time API gateway for ClickHouse — schema-aware ingest, async batching, real-time streaming, and tiered query caching in a single binary.","codeRepository":"https://github.com/Wave-RF/WaveHouse","programmingLanguage":["Go","TypeScript"],"license":"https://opensource.org/license/apache-2.0/","image":"https://wavehouse.dev/og.png","url":"https://wavehouse.dev","author":{"@id":"https://wave-rf.com/#org"}}]} + {"@context":"https://schema.org","@graph":[{"@type":"Organization","@id":"https://wave-rf.com/#org","name":"Wave RF","url":"https://wave-rf.com","email":"hello@wave-rf.com","sameAs":["https://github.com/Wave-RF"]},{"@type":"SoftwareSourceCode","@id":"https://wavehouse.dev/#project","name":"WaveHouse","description":"The open-source real-time API gateway for ClickHouse — schema-aware ingest, async batching, real-time streaming, and tiered query caching in one binary.","codeRepository":"https://github.com/Wave-RF/WaveHouse","programmingLanguage":["Go","TypeScript"],"license":"https://opensource.org/license/apache-2.0/","image":"https://wavehouse.dev/og.png","url":"https://wavehouse.dev","author":{"@id":"https://wave-rf.com/#org"}}]} hero: title: WaveHouse - tagline: The open-source real-time API gateway for ClickHouse. Schema-aware ingest, async batching, real-time streaming, and tiered query caching — in a single binary. + tagline: The open-source real-time API gateway for ClickHouse. Schema-aware ingest, async batching, real-time streaming, and tiered query caching — in one binary. actions: - text: Read the docs link: /getting-started @@ -220,7 +220,7 @@ Self-hosting WaveHouse is deliberately boring — one binary, one dependency. Bu

WaveHouse is alpha — and built entirely in the open.

-

Apache-2.0-licensed, single binary, no vendor lock-in — self-host it forever, or let us run it on WaveHouse Cloud. We ship with honest expectations — see the support cadence and security policy. Kick the tires and tell us where it breaks.

+

Apache-2.0-licensed, one binary, no vendor lock-in — self-host it forever, or let us run it on WaveHouse Cloud. We ship with honest expectations — see the support cadence and security policy. Kick the tires and tell us where it breaks.

Get started in five minutes Star on GitHub diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index 18bc714e..d062fb43 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -15,13 +15,12 @@ It is deliberately detailed: this is a hot, concurrency-heavy path, and the goro | --- | --- | | `worker.go` | `StartIngestWorker`, the `dispatchLoop`, `parseMsg` (+ `rejectPoison` for an envelope it cannot read), the per-tenant-table `tableBatcher`/`tableLoop`, `flushTable` (splits a batch per column list via `groupByColumns`, or hands the whole batch of a tenant with no ClickHouse connection to `parkBatch`) and `flushGroup` (bulk insert with a row-by-row isolation fallback when ClickHouse rejects the batch, or refuses a multi-row batch for its size — `chconn.Splittable`), `retryLater` (a batch ClickHouse could not take, handed back for a delayed redelivery), `insertToClickHouse` (into the batch's tenant's ClickHouse, `chconn.Pools.Target`), `handleSuccess` (acks, after `invalidate` bumps the tenant's cache namespaces — under every tenant on the same ClickHouse address and database, through the cache `internal/app` hands the worker, since they read the same tables), `sendToDLQ`/`parkOnDLQ` | | `backoff.go` | The retry backoff per ClickHouse pool: one outage backs off every table on it together, probing once per window; a failure of one table (read-only, too many parts or mutations, a missing grant) backs off that table alone — see [When ClickHouse cannot take an insert](#when-clickhouse-cannot-take-an-insert) | -| `compact.go` | `EncodeCompactRow` — renders one record as a `JSONCompactEachRow` line over the table's **insertable** columns, in declaration order. Serialization only: it validates nothing and judges no value | | `sweeper.go` | The **Active Sweeper** — every minute, asks the MQ to purge the events that are both written to ClickHouse and past the SSE gap window (the purge arithmetic below lives in `internal/mq/purge.go`) | | `claims.go` | `ClaimShards` — over the external broker's shards, narrows the worker to this process's share: the membership lease, the per-tick assignment, handover by drain then release, the reset of a dead owner's shards at takeover, each shard's share of the `MaxHeld` budget of rows held, the halt that lets the worker write what it holds before a clean stop releases its shards, and the shard gauges. The behavior is in [Scaling ingest processes](/deployment#scaling-ingest-processes). | | `assign.go` | `assignUnits` — rendezvous hashing capped at ⌈units/slots⌉, the same owners in every process for the same view | | `types.go` | `EventMessage` wire format and the `BufferConsumerName` constant | -The pipeline is **insert-only**. (Upgrading across the v2 envelope? [Drain the queue first](/deployment#upgrading-across-the-v2-ingest-envelope).) The wire format carries `{table_name, scope, received_timestamp, format, columns, row}`: `row` is one `JSONCompactEachRow` line — a positional JSON array — and `columns` names its positions — the table's insertable columns, in declaration order (a `MATERIALIZED` or `ALIAS` column cannot be named in an `INSERT`, so it is not part of the row's contract). (`scope` is reserved and always `""` today.) Each NATS message is its own envelope, so the names ride along per record; where they are carried once is the `INSERT` the worker emits per group. The worker parses the envelope, groups a batch by column list, and bulk-`INSERT`s each group as `INSERT INTO … (cols) FORMAT JSONCompactEachRow` — schema validation already happened at the HTTP ingest handler, before publish. Non-insert mutations go through `POST /v1/ops/query` (admin-only) or an operator-authored [pipe that writes](/pipes#pipes-that-write). +The pipeline is **insert-only**. (Upgrading across the v2 envelope? [Drain the queue first](/deployment#upgrading-across-the-v2-ingest-envelope).) Each NATS message is one envelope — `{table_name, scope, received_timestamp, format, columns, row}`, documented field by field in [API → Internal Wire Format](/api#internal-wire-format-nats). What matters here: `row` is not something WaveHouse encodes, it is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced at ingest time, through the chtypes engine, and `columns` names its positions. The worker parses the envelope, groups a batch by column list, and bulk-`INSERT`s each group as `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with the settings chtypes compiled the row with plus `async_insert=0`. A role that may not write every column produces a shorter `columns` list, which is simply another batch group. Non-insert mutations go through `POST /v1/ops/query` (admin-only) or an operator-authored [pipe that writes](/pipes#pipes-that-write). ## High-level shape @@ -59,12 +58,12 @@ flowchart LR Note the embedded broker's stream is **dual-use**: it is both the durable buffer feeding the worker and the replay buffer that SSE clients gap-fill from. That is why a custom sweeper exists there instead of plain work-queue auto-deletion; `mq.backend: nats` splits the two roles instead, with work-queue partitions and a separate history stream (see [Scaling out](#scaling-to-multiple-instances)). -:::note[Omitted columns on `Nullable` columns with a default] -Inserts also pin `input_format_null_as_default=1`. A positional row has one value per insertable column and no way to say "absent", so a field the record omitted rides as an explicit `null` in its slot. That setting turns the `null` back into the column's default for a **non-nullable** column, matching what omitting the key did under `JSONEachRow` — but on a `Nullable(T) DEFAULT …` column ClickHouse stores `NULL` whatever the setting says, because only an *absent* key ever took the default. So such a column now stores `NULL` where it previously took its default. Verified on ClickHouse 26.6.3. +:::note[Omitted columns take their real DEFAULT, not `null`] +The batch that reaches `insertToClickHouse` is not assembled from the request body — it is the bytes `Table.Ingest` returned for each accepted record, produced by ClickHouse's own writer. An omitted field's `DEFAULT` (or the type's implicit zero) was evaluated before that line existed, so a `Nullable(T) DEFAULT …` column takes its default exactly as an `INSERT` naming fewer columns would. Verified on ClickHouse 26.6.3. ::: -:::note[ClickHouse timestamp parsing] -Inserts pin `date_time_input_format=best_effort` — the server default since ClickHouse 26.5, but on older servers the `basic` default rejects the canonical RFC 3339 form's `Z` suffix ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). The ordinary spellings (zone-less date-times, 9–10-digit Unix-seconds strings) parse identically under both settings. (This is moot for anything an older build buffered: the upgrade deletes it — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope).) Bare digit-strings of other lengths are the exception: `best_effort` reads them as ClickHouse's calendar/epoch shapes, where `basic` read a plain `DateTime` column's digit string of five or more digits as Unix seconds (shorter runs it rejected outright, where `best_effort` reads `"2026"` as a year): under `best_effort` `"20260711"` stores 2026-07-11, where `basic` stored 1970-08-23. `DateTime64` columns diverge the same way on calendar-shaped runs, and additionally whenever an epoch run's unit doesn't match the column scale (under `basic`, runs longer than 10 digits are ticks at the column's own scale; `best_effort` unit-detects 13/16/19-digit runs as ms/µs/ns). A producer relying on the old `basic` reading changes meaning as soon as this WaveHouse version is deployed — the pin, not a ClickHouse upgrade, is what flips the parse. +:::note[Insert settings pinned] +Inserts pin the same parsing settings chtypes compiled the row with — `date_time_input_format=best_effort` and `input_format_null_as_default=1` — plus `async_insert=0`, which the worker adds itself (not a parsing setting, so it's never passed to chtypes): unpinned, ClickHouse 26.2+'s server-default async insert costs a flush-wait floor per statement and loses per-row error attribution (`While executing WaitForAsyncInsert` instead of `(at row N)`). `deduplicate_insert`'s server default (block-hash dedup, `enable` from 26.2 for plain `MergeTree`) is deliberately left unpinned — fine for retried-identical-batch at-least-once delivery, but worth knowing about if the worker's retry classifier ever needs to distinguish it from two legitimately identical batches. ::: ## The journey of one event diff --git a/docs/src/content/docs/reverse-proxy.mdx b/docs/src/content/docs/reverse-proxy.mdx index 408bc8d2..5651d531 100644 --- a/docs/src/content/docs/reverse-proxy.mdx +++ b/docs/src/content/docs/reverse-proxy.mdx @@ -104,7 +104,7 @@ Set your own **outer** limit at the proxy, sized to your real needs: The effective limit is the smaller of the proxy's and WaveHouse's. For ingest, WaveHouse's 16 MiB is the ceiling — raising the proxy above it won't help, because the server rejects first. The cap applies to **every** body shape, NDJSON included: WaveHouse reads the whole body before parsing it, so a line-framed batch is bounded by the same 16 MiB cap as a JSON array. Split an upload larger than the cap across several requests. See [Batch Ingest](/api#batch-ingest). :::note[Size the container for concurrency, not just for one request] -Ingest reads the whole body into memory before parsing it, so peak memory per in-flight request tracks the **body**, not one record — and it is a multiple of the body. `bytes.Buffer` grows by doubling to the next power of two and then once more to probe for EOF, so any body above ~16 MiB − 512 B lands in a **32 MiB** allocation (a 9 MiB body already costs 16 MiB), and that final doubling *copies*: the old 16 MiB array and the new 32 MiB one are both live while it runs, a transient of roughly **48 MiB — 3× the body — for a single request**. Hitting the cap does not avoid it, since a rejected over-cap body allocates the same 32 MiB before the `413`. On top of that sits the decoded Go value of the record in flight; ingest still decodes record by record, so for a batch of small rows that term is minor. A single-object body **is** one record, and the decode runs *before* schema validation: the body is materialized as a `map[string]any` carrying whatever keys it arrived with, and only then is an unknown column rejected. So a flat object of many tiny keys is fully in memory before anything bounds it — a 16 MiB body of ~1.4M one-byte values decodes to well over 100 MiB, which *is* the order-of-magnitude amplification quoted above. (A single oversized array element or NDJSON line costs the same, bounded by 16 MiB and 10 MiB respectively.) The 3× figure covers the buffering term alone. Nothing bounds the *total* across concurrent uploads either, so size for concurrent uploads × **at least 10×** the largest body you accept — an array-valued key amplifies harder still, around 18× measured, and nesting goes higher — or lower the proxy's body cap for `/v1/ingest` below 16 MiB. Size the container memory limit and the proxy's body cap together, and cap concurrency at the proxy if you accept large batches. A server-side bound on total in-flight bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). +Ingest reads the whole body into memory before parsing it, so peak memory per in-flight request tracks the **body**, not one record — and it is a multiple of the body. `bytes.Buffer` grows by doubling to the next power of two and then once more to probe for EOF, so any body above ~16 MiB − 512 B lands in a **32 MiB** allocation (a 9 MiB body already costs 16 MiB), and that final doubling *copies*: the old 16 MiB array and the new 32 MiB one are both live while it runs, a transient of roughly **48 MiB — 3× the body — for a single request**. Hitting the cap does not avoid it, since a rejected over-cap body allocates the same 32 MiB before the `413`. WaveHouse no longer decodes a record into Go values, so the old per-record amplification is gone; what sits on top of the buffering term now is ClickHouse's own parse of those bytes and the rows it renders back. Nothing bounds the *total* across concurrent uploads, so size the container memory limit and the proxy's body cap together, and cap concurrency at the proxy if you accept large batches. A server-side bound on total in-flight bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). ::: ## Server-Sent Events (SSE) @@ -201,7 +201,7 @@ These limit the *whole* request regardless of traffic, so no keepalive extends t WaveHouse does **not** derive a client IP from forwarded headers — it does no per-IP logic (rate limiting and IP allow/deny are the proxy's job) and does not trust `X-Forwarded-For` / `X-Real-IP` / `True-Client-IP` to rewrite the connection's source address. So a forged forwarded header has no effect on WaveHouse, and `r.RemoteAddr` (what OpenTelemetry records as the peer) is the honest immediate peer — your proxy, when one is in front. Still, don't expose `:8080` to untrusted clients: bind WaveHouse to a private interface or firewall the port so the proxy is the only path in. Capturing the real client IP in WaveHouse's own traces and logs — trusted-proxy-aware, so it can't be spoofed — is tracked in [#333](https://github.com/Wave-RF/WaveHouse/issues/333). ::: -- **`Content-Type`** — forward it **verbatim**; ingest reads the format from it and refuses anything it cannot read ([details](/api#post-v1ingesttabletable--ingest-data)). Appending rather than replacing is safe *only* if the proxy sends a second header **line** — those are resolved together and accepted when they agree. A proxy that **merges** duplicates into one comma-joined value (Envoy's `append: true`, and some WAF rewrite rules) produces `application/json, application/json`, which is a `415` on every ingest **even though both halves agree**: a comma-joined value is refused unless the value as a whole still parses as one media type. Replacing it is worse than merging, because it fails silently: an NDJSON batch declared `application/json` is read as the single object it starts with and the remaining lines are dropped behind a `200` ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)) — no error to alert on. If ingest starts returning `415` fleet-wide after a proxy change, look here first. +- **`Content-Type`** — forward it **verbatim**; ingest reads the format from it and refuses anything it cannot read ([details](/api#post-v1ingesttabletable--ingest-data)). Appending rather than replacing is safe *only* if the proxy sends a second header **line** — those are resolved together and accepted when they agree. A proxy that **merges** duplicates into one comma-joined value (Envoy's `append: true`, and some WAF rewrite rules) produces `application/json, application/json`, which is a `415` on every ingest **even though both halves agree**. Replacing it is worse than merging, because it fails silently: an NDJSON batch declared `application/json` is read as the single object it starts with and the remaining lines are dropped behind a `200` ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)) — no error to alert on. And a proxy that rewrites `text/csv` or `text/tab-separated-values` (including their `; header=present` variants, which select the header-line formats) will turn a valid positional batch into a `415` or, if it drops the parameter, into a different reading of the header line (a rejected record under `header=absent`, a silently consumed header on a bare type). If ingest starts returning `415` fleet-wide after a proxy change, look here first. - **CORS** — WaveHouse applies its own CORS from the settings directory's `cors.allowed_origins` — over a nested directory, [the list of the tenant the request names](/deployment#multi-tenant-deployments), the preflight following the `X-Tenant-ID` the proxy stamps on `OPTIONS`. Let one layer own CORS: either pass it through the proxy untouched (recommended), or strip it from WaveHouse and do it at the proxy — not both, or browsers see duplicate `Access-Control-Allow-Origin` headers and reject the response. diff --git a/docs/src/content/docs/sdk/queries.md b/docs/src/content/docs/sdk/queries.md index 38de8890..0bc79b1c 100644 --- a/docs/src/content/docs/sdk/queries.md +++ b/docs/src/content/docs/sdk/queries.md @@ -40,9 +40,9 @@ const { data } = await clicks.insert([ // data: { ok, total, succeeded, failed, duplicates, results? } ``` -For an array insert, `data.ok` is `true` only when every record succeeded (`failed === 0`). Inspect `data.failed` and `data.results` (each `{ index, ok|duplicate|error }`, 1-based `index`) for partial failures — the call's top-level `error` is reserved for whole-request failures (network, `404` unknown table, `403` forbidden, `503` backpressure). An empty array is a no-op and sends no request. The array path sends one request regardless of size, so it is bound by the same 16 MiB [request-body cap](/reverse-proxy#request-body-size-limits); bounded-concurrency chunking of very large arrays is tracked in [#196](https://github.com/Wave-RF/WaveHouse/issues/196). +For an array insert, `data.ok` is `true` only when every record succeeded (`failed === 0`). Inspect `data.failed` and `data.results` (each `{ index, ok|duplicate|error, exception_code? }`, 1-based `index`, `exception_code` being ClickHouse's own error code on a parser rejection) for partial failures — the call's top-level `error` is reserved for whole-request failures (network, `404` unknown table, `403` forbidden, `503` backpressure). An empty array is a no-op and sends no request. The array path sends one request regardless of size, so it is bound by the same 16 MiB [request-body cap](/reverse-proxy#request-body-size-limits); bounded-concurrency chunking of very large arrays is tracked in [#196](https://github.com/Wave-RF/WaveHouse/issues/196). -> `POST /v1/ingest` also accepts a raw JSON array or a single object directly, so non-SDK clients can send whichever shape is convenient — but the `Content-Type` is **required** and decides the format (`application/json` or `application/x-ndjson`); a request without one is rejected with `415`. The SDK always sets it. See the [API reference](/api#post-v1ingesttabletable--ingest-data). +> `POST /v1/ingest` also accepts a raw JSON array or a single object directly, and positional `text/csv` / `text/tab-separated-values` bodies (add `; header=present` to send a header line of column names, or `; header=absent` to turn ClickHouse's header auto-detection off), so non-SDK clients can send whichever shape is convenient — but the `Content-Type` is **required** and decides the format; a request without one is rejected with `415`. The SDK always sets it. See the [API reference](/api#post-v1ingesttabletable--ingest-data). ### `.insertNDJSON(source, opts?)` diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index 9d0a9ba7..f17f64fc 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -147,7 +147,7 @@ Codegen reads `/v1/ops/schema`, which is **admin-only**. Against a non-dev serve | `--out`, `-o` | Output .d.ts file path | `./wavehouse.d.ts` | | `--auth`, `-a` | Bearer token (if auth required) | — | -The generated row type is the **read** shape — with one exception running the other way: an `EPHEMERAL` column declares a default, so codegen emits it too, yet no query can ever return it. There the type says readable where only the write is real. A `MATERIALIZED` or `ALIAS` column declares a default, so it is emitted as optional — but supplying one on `insert` is a `400` (`column "x" of table "t" is materialized and cannot be inserted`), and the type will not catch it. Omit computed columns; the server fills them in. +The generated row type is the **read** shape, and computed columns are where it and the server disagree. An `EPHEMERAL` column declares a default, so codegen emits it, yet no query can ever return it — the type says readable where only the write is real. `MATERIALIZED` and `ALIAS` columns declare defaults too, so they are emitted as optional, but supplying any of the three on `insert` is a `400` carrying ClickHouse's own code 117 (`Unknown field found while parsing JSONEachRow format: x`), and the type will not catch it. Omit computed columns; the server fills them in. **Example output:** @@ -172,7 +172,7 @@ export interface ClicksRow { | ClickHouse Type | TypeScript Type | |----------------|-----------------| | `String`, `FixedString`, `UUID`, `DateTime*`, `Date*`, `Enum*`, `IPv4/6` | `string` | -| `UInt*`, `Int*`, `Float*`, `Decimal*` | `number` | +| `UInt*`, `Int*`, `Float*`, `Decimal*` | `number` — `Decimal*` comes back as a JSON number, not a string | | `Bool` | `boolean` | | `Nullable(T)` | `T \| null` | | `Array(T)` | `T[]` | diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index 27ee3a95..92e91e12 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -97,9 +97,9 @@ interface StreamEvent { } ``` -`data` is a row **object**, as it always has been — but the wire underneath is positional. The server sends the column list in its own `event: schema` frame — before the first row, and again whenever the list drifts on the **live** path (with one exception after a gap-fill, below) — and each row as a JSON array; the SDK keeps the announced list and zips every row against it, so this shape is unchanged and nothing in your code moves. It matters in three places. A column the producer omitted arrives as an explicit `null` rather than an absent key. The row object has a **null prototype**: a ClickHouse column may legitimately be named `__proto__`, and on an ordinary object that assignment hits the inherited setter and the value disappears — so the SDK builds each row with `Object.create(null)`. Property access, spreading, `JSON.stringify` and destructuring all behave normally; what does not is anything inherited from `Object.prototype`, so use `Object.hasOwn(row, "x")` rather than `row.hasOwnProperty("x")`, and don't rely on `` `${row}` `` or `row.constructor`. (`liveQuery`'s REST backfill half still yields ordinary objects.) And a **raw** SSE consumer (a hand-rolled `EventSource`) must do the zipping itself — see [the wire format](/api#get-v1stream--server-sent-events-stream). +`data` is a row **object**, as it always has been — but the wire underneath is positional. The server sends the column list in its own `event: schema` frame — before the first row, and again whenever the list drifts on the **live** path (with one exception after a gap-fill, below) — and each row as a JSON array; the SDK keeps the announced list and zips every row against it, so this shape is unchanged and nothing in your code moves. It matters in two places. The row object has a **null prototype**: a ClickHouse column may legitimately be named `__proto__`, and on an ordinary object that assignment hits the inherited setter and the value disappears — so the SDK builds each row with `Object.create(null)`. Property access, spreading, `JSON.stringify` and destructuring all behave normally; what does not is anything inherited from `Object.prototype`, so use `Object.hasOwn(row, "x")` rather than `row.hasOwnProperty("x")`, and don't rely on `` `${row}` `` or `row.constructor`. (`liveQuery`'s REST backfill half still yields ordinary objects.) And a **raw** SSE consumer (a hand-rolled `EventSource`) must do the zipping itself — see [the wire format](/api#get-v1stream--server-sent-events-stream). A column the producer omitted is no longer `null` on the wire: WaveHouse's ingest validation runs ClickHouse's own parser in-process, which evaluates the column's `DEFAULT` (or its implicit zero value) before the row is published — the same as a native `INSERT` naming fewer columns than the table has. -Row values of top-level `DateTime`/`DateTime64` columns inside `data` (not timestamps nested in `Array`/`Map`/`Tuple` columns) arrive in canonical RFC 3339 UTC (`2026-06-21T04:00:00.123Z`), matching what `/v1/query` returns for the same row — `new Date(value)` parses correctly with no zone fix-up. Values WaveHouse couldn't canonicalize (ingest is fail-open) stream in the producer's original spelling, and the `/v1/query` match doesn't hold for them: a spelling ClickHouse accepted anyway still queries back in canonical UTC (one it rejected never lands in the table at all), and a zone-less date-time is what `new Date()` reads as *local* time — though a date-only `YYYY-MM-DD` string is read as UTC, an ECMAScript quirk (see [Timestamp canonicalization](/api#timestamp-canonicalization)). +Row values of top-level `DateTime`/`DateTime64` columns inside `data` (not timestamps nested in `Array`/`Map`/`Tuple` columns) arrive as ClickHouse itself renders them — `"2026-06-21 04:00:00.123"`, space-separated, no `Z` suffix, in the column's declared zone else the server's default — matching what `/v1/query` returns for the same row byte-for-byte, by construction (see [Timestamp rendering](/api#timestamp-rendering)). `new Date(value)` does **not** parse this form correctly out of the box: ECMAScript's `Date` constructor reads a bare `YYYY-MM-DD HH:MM:SS` string as *local* time, not UTC, so replace the space with `T` and append the zone before parsing, or parse it with a zone-aware library. ### Transport Behavior diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 49975c2c..f5ad93ff 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -188,8 +188,8 @@ What stays in boot config is only what cannot change under a running process — Every per-tenant dedupe knob lives here. Where the seen ids are kept (`dedupe.backend`) and how long a claim is held (`dedupe.lease`) are [boot config](/configuration#dedupe), the same for every tenant. The switch and its fields are resolved per record from one snapshot (table override → global value): - `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens or closes this tenant's store (in the embedded Pebble instance at `/pebble`, or its share of the DynamoDB table under `dedupe.backend: dynamodb`), so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`503 dedupe store unavailable`, `Retry-After: 5`) until it opens — the files asked for dedupe, so publishing un-deduped is not a fallback. With `dedupe.backend: pebble` that is the next reload or restart, and at boot a failed open over a flat directory refuses to start, like every other store; with `dynamodb` it is the background retry described below. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) with `dedupe.backend: pebble`, every tenant's seen ids live in that one instance, each key led by its tenant and table, and it is open while any tenant's switch is on: each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and keeps its seen ids for the folder that restores it; and if that instance fails to open, at boot or on reload, every tenant with dedupe on fails closed — its ingest answers `503 dedupe store unavailable` (`Retry-After: 5`) until a reload opens it — while the tenants with dedupe off carry on. Under `dedupe.backend: dynamodb` the table check plays the instance's part, in either shape: the table is checked whether or not any tenant's switch is on, and a table that fails it fails every tenant with dedupe on closed until the check, retried in the background and at once after every reload, passes. Only a misconfigured table (missing, the wrong key schema, access denied) over a flat directory whose tenant has dedupe on refuses boot instead ([Configuration](/configuration#dynamodb-dedupe)). -- `dedupe.id_field` (seed default `event_id`) — JSON field name in the ingest body used as the dedup key. An id is a duplicate only within its own tenant and table: the same value in two tables is two ids. An id longer than 1,024 bytes once escaped (every byte but an ASCII letter, digit, `_` or `-` takes three) is stored as its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. While its record is being published, an id is held for its lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default): another request carrying the same id meanwhile gets `503` (`a request with the same dedupe id is in flight`) with the lease, in whole seconds, as `Retry-After` — see [the ingest errors](/api#post-v1ingesttabletable--ingest-data). An id is committed only after its record is published; if that commit fails (counted by `wavehouse_ingest_dedupe_commit_failed_total`, which should stay at zero), the record is still answered `ok` and the id lapses with its lease: a retry of it before then answers in-flight, one inside the ingest queue's two-minute duplicate window is dropped there by its idempotency key, and one after that is stored again. -- `dedupe.require_id` (seed default `false`) — controls what happens to a row missing `id_field`, or carrying it as `null` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. +- `dedupe.id_field` (seed default `event_id`) — the **column** whose stored value is the dedup key. It is read out of the row ClickHouse rendered, not out of the request body, so the key is the value that will be stored rather than the caller's spelling (`256` into a `UInt8` keys on `0`) — identical for the documented case, a string id. "Missing" therefore means the row carries no value for it, which is what an omitted `String` column produces; a numeric id column cannot tell an omitted `0` from a supplied one. An id is a duplicate only within its own tenant and table: the same value in two tables is two ids. An id longer than 1,024 bytes once escaped (every byte but an ASCII letter, digit, `_` or `-` takes three) is stored as its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. While its record is being published, an id is held for its lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default): another request carrying the same id meanwhile gets `503` (`a request with the same dedupe id is in flight`) with the lease, in whole seconds, as `Retry-After` — see [the ingest errors](/api#post-v1ingesttabletable--ingest-data). An id is committed only after its record is published; if that commit fails (counted by `wavehouse_ingest_dedupe_commit_failed_total`, which should stay at zero), the record is still answered `ok` and the id lapses with its lease: a retry of it before then answers in-flight, one inside the ingest queue's two-minute duplicate window is dropped there by its idempotency key, and one after that is stored again. +- `dedupe.require_id` (seed default `false`) — controls what happens to a row with no value for `id_field` — omitted, or `null` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. - `dedupe.retention` (optional; seed default `"0"`) — how long a committed id stays a duplicate, as a Go duration string: `"24h"`, `"720h"` (30 days), `"90m"`. There is no day unit. `"0"` keeps every id forever, which was the only behavior before this key existed, and a `config.json` without the key means the same. Once an id's retention has ended, the next record carrying it is published as new. With `dedupe.backend: pebble`, a background sweep over the shared Pebble instance deletes the expired id: first about a minute after the instance opens (when the first tenant switches dedupe on), then hourly while any tenant keeps it on, counted by `wavehouse_dedupe_swept_keys_total{reason="expired"}`. With `dynamodb`, no sweep runs: the table's TTL on `ex` deletes the item ([Deployment](/deployment#a-shared-dedupe-table-on-dynamodb)). A finite retention must be at least `"2m"`, the embedded ingest queue's duplicate window (under `mq.backend: nats`, keep it at least the partitions' `duplicate_window` too, which boot warns about but this file cannot check; see [External NATS](/deployment#create-the-topology)): every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and then dropped by the queue as a copy, while the client was told it was accepted. A retention below that is refused, not raised to the minimum; so are a negative value and anything that is not a duration, such as `"30d"`, a number with no unit (`"300"` needs one: `"300s"`; `"0"` is the one exception), or a JSON number rather than a string. Hot-reloadable: a change applies to ids committed after the reload, and an id already committed keeps the expiry it was stored with. - `dedupe.tables.
.{id_field, require_id, retention}` — per-table overrides; each entry overrides only the fields it names and inherits the rest, so a table with no `retention` keeps the tenant's (forever when the tenant sets none). A table can keep ids for a shorter time than its tenant, or for longer, or forever (`"retention": "0"`) under a finite tenant retention. diff --git a/docs/src/content/docs/why-wavehouse.md b/docs/src/content/docs/why-wavehouse.md index 883d1928..ba08cd60 100644 --- a/docs/src/content/docs/why-wavehouse.md +++ b/docs/src/content/docs/why-wavehouse.md @@ -171,7 +171,7 @@ Where it differs from WaveHouse: | Dimension | Tinybird | WaveHouse | | --------- | -------- | --------- | -| Hosting | SaaS only (managed tiers: Developer $49/mo → Enterprise custom) | Self-host, single binary | +| Hosting | SaaS only (managed tiers: Developer $49/mo → Enterprise custom) | Self-host, one binary + local artifact | | Pricing model | Pay for allocated vCPU/QPS/storage; egress fees for cross-region | Your infra; no per-query or per-GB fee | | Data residency | Their infrastructure | Your infrastructure | | Source of truth for schema | Tinybird datasource definitions | Your ClickHouse tables (`system.columns`) | From e07115894299598ae4f3db738d5a65df62a1c45c Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:40:46 -0400 Subject: [PATCH 03/70] feat(policy): resolved row-filter predicates and a typed WHERE renderer Export the resolved row-filter predicate (Predicate) and hand it out through ResolvedPermissions.Predicates, the same resolution the query path renders, so the stream can evaluate it with ClickHouse's own engine. ResolvedSelect.WhereSQL(colType) renders the clause with an integer column's claims bound through the strict round-trip cast, so a claim that does not fit the column matches nothing instead of wrapping. WhereClause, WhereParams and RowVisible stay until their callers move. chsql gains the shared {p:String} encoding (EscapeStringParam), the strict cast (IntegerType, StrictInt, IntParam), and QuoteIdent now escapes NUL and the control characters exactly as ClickHouse's backQuote does. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/chsql/chsql.go | 146 +++++++++++++++++++++++++++++---- internal/chsql/chsql_test.go | 106 ++++++++++++++++++++++++ internal/policy/policy.go | 103 +++++++++++++++++------ internal/policy/policy_test.go | 138 +++++++++++++++++++++++++++++++ internal/policy/rowfilter.go | 2 +- 5 files changed, 451 insertions(+), 44 deletions(-) diff --git a/internal/chsql/chsql.go b/internal/chsql/chsql.go index e91ba8b5..47d2cb47 100644 --- a/internal/chsql/chsql.go +++ b/internal/chsql/chsql.go @@ -1,16 +1,22 @@ // Package chsql holds ClickHouse SQL helpers shared across packages that build -// SQL — primarily safe identifier quoting. It is dependency-free so both -// internal/query and internal/policy can use it without an import cycle. +// SQL: safe identifier quoting, the encoding a value needs to survive a +// `{p:String}` query parameter, and the strict cast an integer column's claims +// are compared through. It is dependency-free so internal/query, +// internal/policy and internal/typelayer can all use it without an import +// cycle. package chsql import "strings" -// identEscaper escapes a ClickHouse identifier's special characters exactly as -// ClickHouse's own backQuote() does (confirmed against SHOW CREATE TABLE on a -// live server): a backslash becomes `\\` and a backtick becomes “ \` “. The -// two replacements run in a single left-to-right pass, so neither re-processes -// the other's output. -var identEscaper = strings.NewReplacer(`\`, `\\`, "`", "\\`") +// identEscaper is ClickHouse's backQuote() escaping, byte for byte as the +// chtypes library's QuoteIdentifier returns it (measured over all 256 bytes; +// typelayer's parity test pins it): `\\`, `\“, and `\0 \b \t \n \f \r` +// for NUL and those control characters. One left-to-right pass, so no +// replacement re-processes another's output. +var identEscaper = strings.NewReplacer( + `\`, `\\`, "`", "\\`", + "\x00", `\0`, "\b", `\b`, "\t", `\t`, "\n", `\n`, "\f", `\f`, "\r", `\r`, +) // QuoteIdent renders any value as a backtick-quoted ClickHouse identifier // (column, table, or alias). It is the single place an identifier becomes SQL @@ -25,19 +31,125 @@ var identEscaper = strings.NewReplacer(`\`, `\\`, "`", "\\`") // (verified on a live server). Always quoting is unconditionally correct and // needs no keyword table. // -// Values are never quoted here; they remain positional `?` parameters bound by -// the driver. +// A value is never rendered into SQL text here; it binds as a query parameter +// and is encoded by EscapeStringParam. func QuoteIdent(name string) string { return "`" + identEscaper.Replace(name) + "`" } -// BindUnsafe reports whether an identifier contains a character that -// clickhouse-go's positional binder miscounts. Today that is just '?': the -// driver counts every '?' in the query text — even inside a backtick-quoted -// identifier — so a name containing '?' would shift the value parameters that -// follow it. Callers refuse such names (fail closed) rather than risk mis-binding -// a value (including a row-level-security filter value). Pathological; no real -// schema names a column '?'. Tracked in Wave-RF/WaveHouse#279. +// BindUnsafe reports whether an identifier contains a character that the +// positional-placeholder scan miscounts. Today that is just '?': the scan that +// rewrites a built query's `?` placeholders into named `{pN:…}` parameters +// walks the SQL text left to right and cannot tell a placeholder from a '?' +// inside a backtick-quoted identifier, so a name containing '?' would shift +// every value that follows it. Callers refuse such names (fail closed) rather +// than risk mis-binding a value (including a row-level-security filter value). +// Pathological; no real schema names a column '?'. Tracked in +// Wave-RF/WaveHouse#279. func BindUnsafe(name string) bool { return strings.ContainsRune(name, '?') } + +// EscapeStringParam encodes one value for a ClickHouse `{p:String}` query +// parameter — the SINGLE encoding both read surfaces use, so the SQL path +// (internal/query, over the HTTP interface) and the row-filter path +// (internal/typelayer, over the chtypes artifact) cannot disagree about what a +// claim value is. +// +// ClickHouse reads a scalar parameter with its ESCAPED-TEXT reader, so a raw +// backslash starts an escape sequence and a raw tab or newline ends the field. +// Measured on 26.6.3.62 over the HTTP interface: `param_p0=a\b` came back +// holding a backspace with no error, and a raw tab or newline was a hard code +// 457 parse error (a 500 for the caller). Measured on the 26.6 chtypes +// artifact through typelayer's compiled filters: the same stored values were +// answered false (the backslash case) or refused at compile time (tab, +// newline, a trailing backslash), and the same encoding made all of them +// compare equal. Encoding `\` → `\\`, tab → `\t`, newline → `\n`, CR → `\r` +// round-trips every value byte for byte on both surfaces, including an +// embedded NUL. +// +// An Array(String) parameter takes a DIFFERENT rule and must NOT be run +// through this one: its elements are read as QUOTED values, where a raw tab or +// newline rides through untouched and only `'` and `\` need escaping. +// Applying both encodings is corruption — `a\b` becomes `a\\b`. See +// quoteCHElement in internal/query. +var EscapeStringParam = strings.NewReplacer( + `\`, `\\`, + "\t", `\t`, + "\n", `\n`, + "\r", `\r`, +).Replace + +// IntType is one of ClickHouse's integer type names, as IntegerType returns +// it. It is a distinct type so StrictInt can only be handed a name from that +// closed set, never text read off a schema. +type IntType string + +// intTypes is the closed set IntegerType answers from. Bool is UInt8 +// underneath but compares as a boolean, and Enum/Decimal are not integers, so +// none of them is here. +var intTypes = map[string]IntType{ + "UInt8": "UInt8", "UInt16": "UInt16", "UInt32": "UInt32", "UInt64": "UInt64", + "UInt128": "UInt128", "UInt256": "UInt256", + "Int8": "Int8", "Int16": "Int16", "Int32": "Int32", "Int64": "Int64", + "Int128": "Int128", "Int256": "Int256", +} + +// IntegerType reports whether colType — a ClickHouse type as system.columns +// and the chtypes library spell it — is an integer type, possibly wrapped in +// Nullable(...) and/or LowCardinality(...), and returns the bare integer type. +// +// It only picks which expression form a policy claim is compared through +// (StrictInt for an integer column, a plain {p:String} for everything else); +// it models no ClickHouse semantics. The wrappers are stripped because +// accurateCastOrNull to a LowCardinality type is refused by the server (code +// 455) and the bare type answers identically on a Nullable column (measured +// on 26.6.3.62 and the 26.6 chtypes artifact). A type it does not recognise +// keeps the {p:String} form. +func IntegerType(colType string) (IntType, bool) { + t := colType + for { + inner, ok := unwrap(t, "Nullable(") + if !ok { + inner, ok = unwrap(t, "LowCardinality(") + } + if !ok { + break + } + t = inner + } + it, ok := intTypes[t] + return it, ok +} + +func unwrap(t, prefix string) (string, bool) { + if strings.HasPrefix(t, prefix) && strings.HasSuffix(t, ")") { + return t[len(prefix) : len(t)-1], true + } + return "", false +} + +// StrictInt renders the round-trip strict cast an integer column compares a +// claim bound as {param:String} against: +// +// if(toString(accurateCastOrNull({p:String}, 'T')) = {p:String}, accurateCastOrNull({p:String}, 'T'), NULL) +// +// A bare {p:String} wraps a value at or past 2^64 on every integer column (and +// a 128/256-bit column at its own width), and accurateCastOrNull alone still +// wraps on [U]Int128/[U]Int256. The round trip turns every value that is not +// the canonical spelling of an in-range integer into NULL, which no operator +// admits, while an in-range canonical value compares exactly as before and +// keeps the primary key in use. Measured identical on ClickHouse 26.6.3.62 and +// the 26.6/25.8 chtypes artifacts. +func StrictInt(param string, t IntType) string { + cast := "accurateCastOrNull({" + param + ":String}, '" + string(t) + "')" + return "if(toString(" + cast + ") = {" + param + ":String}, " + cast + ", NULL)" +} + +// IntParam is a positional value the query builder binds through StrictInt +// instead of as a bare {pN:String}: one parameter, referenced from the +// expression the placeholder expands to. +type IntParam struct { + Value string + Type IntType +} diff --git a/internal/chsql/chsql_test.go b/internal/chsql/chsql_test.go index df121cc6..b732a943 100644 --- a/internal/chsql/chsql_test.go +++ b/internal/chsql/chsql_test.go @@ -19,6 +19,7 @@ func TestQuoteIdent(t *testing.T) { {"embedded backslash", `back\slash`, "`back\\\\slash`"}, {"backslash then backtick", "x\\`y", "`x\\\\\\`y`"}, {"star is just a name here", "*", "`*`"}, + {"control characters as backQuote spells them", "a\x00\b\t\n\f\rb", "`a" + `\0\b\t\n\f\r` + "b`"}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { @@ -43,3 +44,108 @@ func TestBindUnsafe(t *testing.T) { } } } + +// TestEscapeStringParam pins the encoding both {p:String} surfaces depend on. +// The single quote and '%'/'_' are deliberately NOT escaped: the value is read +// as an escaped FIELD, not as a quoted literal and not as a LIKE pattern, so +// encoding them would bind characters the caller never wrote. +func TestEscapeStringParam(t *testing.T) { + t.Parallel() + tests := []struct { + name string + in string + want string + }{ + {"plain", "acme", "acme"}, + {"empty", "", ""}, + {"backslash", `a\b`, `a\\b`}, + {"trailing backslash", `trail\`, `trail\\`}, + {"tab", "a\tb", `a\tb`}, + {"newline", "a\nb", `a\nb`}, + {"carriage return", "a\rb", `a\rb`}, + {"single quote is untouched", "O'Brien", "O'Brien"}, + {"like metacharacters are untouched", "50%_off", "50%_off"}, + {"nul is untouched", "a\x00b", "a\x00b"}, + {"backslash then t is not re-read as a tab", `a\tb`, `a\\tb`}, + {"every byte at once", "a\\\tb\nc\rd", `a\\\tb\nc\rd`}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + if got := EscapeStringParam(tt.in); got != tt.want { + t.Errorf("EscapeStringParam(%q) = %q, want %q", tt.in, got, tt.want) + } + }) + } +} + +func TestIntegerType(t *testing.T) { + t.Parallel() + tests := []struct { + in string + want IntType + wantOK bool + }{ + {"UInt8", "UInt8", true}, + {"UInt16", "UInt16", true}, + {"UInt32", "UInt32", true}, + {"UInt64", "UInt64", true}, + {"UInt128", "UInt128", true}, + {"UInt256", "UInt256", true}, + {"Int8", "Int8", true}, + {"Int16", "Int16", true}, + {"Int32", "Int32", true}, + {"Int64", "Int64", true}, + {"Int128", "Int128", true}, + {"Int256", "Int256", true}, + {"Nullable(UInt64)", "UInt64", true}, + {"LowCardinality(UInt32)", "UInt32", true}, + {"LowCardinality(Nullable(Int64))", "Int64", true}, + {"Nullable(Int256)", "Int256", true}, + + {"String", "", false}, + {"LowCardinality(String)", "", false}, + {"Nullable(String)", "", false}, + {"Bool", "", false}, + {"Nullable(Bool)", "", false}, + {"Decimal(18, 4)", "", false}, + {"Decimal64(4)", "", false}, + {"Float64", "", false}, + {"Enum8('a' = 1, 'b' = 2)", "", false}, + {"Enum16('x' = 1)", "", false}, + {"UUID", "", false}, + {"Date", "", false}, + {"DateTime", "", false}, + {"DateTime64(3, 'UTC')", "", false}, + {"IPv4", "", false}, + {"Array(UInt64)", "", false}, + {"Map(String, UInt64)", "", false}, + {"Tuple(UInt64)", "", false}, + {"FixedString(8)", "", false}, + // Not ClickHouse's canonical spelling, so not recognised: the column + // keeps the plain {p:String} form rather than a guessed one. + {"uint64", "", false}, + {"BIGINT UNSIGNED", "", false}, + {"Nullable(UInt64", "", false}, + {"Nullable()", "", false}, + {"", "", false}, + } + for _, tt := range tests { + t.Run(tt.in, func(t *testing.T) { + t.Parallel() + got, ok := IntegerType(tt.in) + if got != tt.want || ok != tt.wantOK { + t.Errorf("IntegerType(%q) = (%q, %v), want (%q, %v)", tt.in, got, ok, tt.want, tt.wantOK) + } + }) + } +} + +func TestStrictInt(t *testing.T) { + t.Parallel() + got := StrictInt("p3", "UInt64") + want := "if(toString(accurateCastOrNull({p3:String}, 'UInt64')) = {p3:String}, accurateCastOrNull({p3:String}, 'UInt64'), NULL)" + if got != want { + t.Errorf("StrictInt = %q\nwant %q", got, want) + } +} diff --git a/internal/policy/policy.go b/internal/policy/policy.go index 3ad4c9c1..61027a39 100644 --- a/internal/policy/policy.go +++ b/internal/policy/policy.go @@ -118,10 +118,11 @@ type ResolvedSelect struct { WhereClause string WhereParams []any // rowFilter is the same row-level-security predicate as WhereClause/WhereParams, - // kept in resolved form so the stream path can evaluate it in memory (RowVisible) - // while the query path renders it to SQL. Both derive from one resolvePredicates - // call in Evaluate, so the two read surfaces can't drift. See rowfilter.go. - rowFilter []resolvedPredicate + // kept in resolved form so the stream path can evaluate it against the row + // (Predicates, RowVisible) while the query path renders it to SQL (WhereSQL). + // Both derive from one resolvePredicates call in Evaluate, so the two read + // surfaces can't drift (#457). + rowFilter []Predicate AllowedAggregations []string DeniedAggregations []string MaxRows int @@ -273,7 +274,7 @@ func evaluateSelect(perms *SelectPermissions, claims map[string]any) *ResolvedPe // (ResolvedPermissions.RowVisible). preds := resolvePredicates(perms.Filter, claims) resolved.Select.rowFilter = preds - clauses, params := predicatesToSQL(preds) + clauses, params := predicatesToSQL(preds, nil) if len(clauses) > 0 { resolved.Select.WhereClause = strings.Join(clauses, " AND ") resolved.Select.WhereParams = params @@ -340,15 +341,20 @@ func evaluateInsert(perms *InsertPermissions, claims map[string]any) *ResolvedPe return resolved } -// resolvedPredicate is one row-filter or check comparison with its claim templates -// already resolved to concrete string values — the shared, render-agnostic form the query -// path turns into SQL (predicatesToSQL) and the stream path evaluates in memory -// (RowVisible). Op is one of "=", "!=", ">", "<", "in". Values holds one element -// for the scalar operators and zero-or-more for "in"; an EMPTY Values matches no -// rows on either surface — an empty/unresolvable "in" set, or a scalar whose -// constant was unresolvable (an absent/null claim, a structured value, or one with -// no canonical form — see resolveTemplate/CanonicalScalar). -type resolvedPredicate struct { +// Predicate is one row-filter comparison with its claim templates already +// resolved to concrete string values — the shared, render-agnostic form the query +// path turns into SQL (predicatesToSQL) and the stream path evaluates against the +// row. Op is one of "=", "!=", ">", "<", "in". Values holds one element for the +// scalar operators and zero-or-more for "in"; an EMPTY Values matches no rows on +// either surface — an empty/unresolvable "in" set, or a scalar whose constant was +// unresolvable (an absent/null claim, a structured value, or one with no +// canonical form — see resolveTemplate/CanonicalScalar). +// +// It is exported because the stream path evaluates it outside this package +// (Predicates). The values are bound as typed parameters there, exactly as they +// are bound as query parameters here — neither surface ever splices one into +// expression text. +type Predicate struct { Column string Op string Values []string @@ -357,17 +363,17 @@ type resolvedPredicate struct { // resolvePredicates resolves each filter's claim templates once into predicates. // Both read surfaces derive from this single result so they can't drift; the // operator order within a column (=, !=, >, <, in) mirrors the former inline SQL. -func resolvePredicates(filters map[string]Filter, claims map[string]any) []resolvedPredicate { - var preds []resolvedPredicate +func resolvePredicates(filters map[string]Filter, claims map[string]any) []Predicate { + var preds []Predicate // An unresolvable constant (ok=false from resolveTemplate) yields a predicate // with NO values, which matches no rows on either surface (#385) — never a // synthesized stand-in that could match some other principal's rows. - scalar := func(col, op, tmpl string) resolvedPredicate { + scalar := func(col, op, tmpl string) Predicate { v, ok := resolveTemplate(tmpl, claims) if !ok { - return resolvedPredicate{Column: col, Op: op} + return Predicate{Column: col, Op: op} } - return resolvedPredicate{Column: col, Op: op, Values: []string{v}} + return Predicate{Column: col, Op: op, Values: []string{v}} } for col, f := range filters { if f.Eq != nil { @@ -383,14 +389,29 @@ func resolvePredicates(filters map[string]Filter, claims map[string]any) []resol preds = append(preds, scalar(col, "<", *f.Lt)) } if f.In != nil { - preds = append(preds, resolvedPredicate{col, "in", toStrings(resolveInValues(*f.In, claims))}) + preds = append(preds, Predicate{col, "in", toStrings(resolveInValues(*f.In, claims))}) } } return preds } -// predicatesToSQL renders resolved predicates into WHERE clauses and bound params. -func predicatesToSQL(preds []resolvedPredicate) ([]string, []any) { +// WhereSQL renders the row filter for the query path: the AND-joined clause +// ("" when the role has no row filter) and its positional `?` params, ready to +// splice into the builder's WHERE. colType returns a column's ClickHouse type +// ("" when unknown, or colType nil): a claim compared against an integer +// column binds as a chsql.IntParam — the strict cast, so a claim that does not +// fit the column matches nothing instead of wrapping — and every other value +// binds as a plain string. +// +// A nil receiver panics, deliberately: see ResolvedPermissions. +func (s *ResolvedSelect) WhereSQL(colType func(column string) string) (string, []any) { + clauses, params := predicatesToSQL(s.rowFilter, colType) + return strings.Join(clauses, " AND "), params +} + +// predicatesToSQL renders resolved predicates into WHERE clauses and bound +// params. colType, when non-nil, picks the strict integer binding (WhereSQL). +func predicatesToSQL(preds []Predicate, colType func(string) string) ([]string, []any) { var clauses []string var params []any for _, p := range preds { @@ -398,6 +419,12 @@ func predicatesToSQL(preds []resolvedPredicate) ([]string, []any) { // caller columns, so a row-filter on a weird-but-legal column name (dots, // spaces, keywords) is emitted safely. qcol := chsql.QuoteIdent(p.Column) + bind := func(v string) any { return v } + if colType != nil { + if it, ok := chsql.IntegerType(colType(p.Column)); ok { + bind = func(v string) any { return chsql.IntParam{Value: v, Type: it} } + } + } switch p.Op { case "in": if len(p.Values) == 0 { @@ -406,10 +433,14 @@ func predicatesToSQL(preds []resolvedPredicate) ([]string, []any) { // fail-open). `IN ()` is not valid SQL, so emit a constant false. clauses = append(clauses, "1 = 0") } else { + // One scalar parameter per element, not one Array(String): the + // strict cast applies per element, and `c IN (E(p0), E(p1))` keeps + // the primary key where `c IN arrayMap(…)` reads every granule + // (measured on 26.6.3.62). placeholders := strings.TrimSuffix(strings.Repeat("?,", len(p.Values)), ",") clauses = append(clauses, fmt.Sprintf("%s IN (%s)", qcol, placeholders)) for _, v := range p.Values { - params = append(params, v) + params = append(params, bind(v)) } } default: @@ -423,7 +454,7 @@ func predicatesToSQL(preds []resolvedPredicate) ([]string, []any) { continue } clauses = append(clauses, fmt.Sprintf("%s %s ?", qcol, p.Op)) - params = append(params, p.Values[0]) + params = append(params, bind(p.Values[0])) } } return clauses, params @@ -432,11 +463,11 @@ func predicatesToSQL(preds []resolvedPredicate) ([]string, []any) { // resolveFilters converts filter definitions with claim templates into SQL WHERE // clauses. Retained as the predicates→SQL composition the query-path tests target. func resolveFilters(filters map[string]Filter, claims map[string]any) ([]string, []any) { - return predicatesToSQL(resolvePredicates(filters, claims)) + return predicatesToSQL(resolvePredicates(filters, claims), nil) } // toStrings normalizes resolveInValues' []any (already canonical strings) to the -// []string a resolvedPredicate carries. +// []string a Predicate carries. func toStrings(vals []any) []string { if len(vals) == 0 { return nil @@ -603,6 +634,26 @@ func (rp *ResolvedPermissions) IsColumnAllowed(col string, insert bool) bool { return false } +// Predicates returns the resolved row-filter predicates for the read side — the +// SAME slice WhereSQL renders into the query's WHERE clause, so the stream and +// the query answer off one resolution and cannot drift (#457). The caller +// evaluates them against the stored row; every value is bound, never spliced. +// +// ok is false when no row may be admitted at all: a denied grant, or one +// resolved for INSERT whose empty read side would otherwise read as "no +// predicates, everything visible" — the same fail-closed shape as CheckClauses. +// A nil receiver means no policy applies, so there is nothing to filter by and +// ok is TRUE with no predicates. +func (rp *ResolvedPermissions) Predicates() ([]Predicate, bool) { + if rp == nil { + return nil, true + } + if !rp.Allowed || rp.Select == nil { + return nil, false + } + return rp.Select.rowFilter, true +} + // CheckClauses returns the insert side's check clauses, and false when the // insert side was never resolved. // diff --git a/internal/policy/policy_test.go b/internal/policy/policy_test.go index f45c4e20..0362a56d 100644 --- a/internal/policy/policy_test.go +++ b/internal/policy/policy_test.go @@ -9,6 +9,8 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/chsql" ) // ptr returns a pointer to v — the Filter operator fields are *string so a @@ -1703,3 +1705,139 @@ func TestEvaluate_OperatorLessFilterAndCheckDenyFailClosed(t *testing.T) { insPerms := Evaluate(ins, "writer", "clicks", "insert", nil) assert.False(t, insPerms.Allowed, "an operator-less check must deny, not drop the rule") } + +// TestPredicates_UnresolvableClaim_NoValuesOnBothPaths pins the #457 fail-closed +// rule on BOTH read surfaces at once, through the accessors the type layer +// reads: a filter template whose claim the token doesn't carry renders the +// constant-false predicate on the query path (WhereSQL) AND yields a predicate +// with NO values on the stream path (Predicates), which the type layer +// refuses without compiling anything. One Evaluate resolution drives both, so a +// claim-less token can never see zero rows on /v1/query yet every row on +// /v1/stream. HasRowFilter must stay true for the failed predicate — dropping it +// would put the role back on the unfiltered once-per-role fast path, the exact +// fail-open this test exists to prevent. +func TestPredicates_UnresolvableClaim_NoValuesOnBothPaths(t *testing.T) { + t.Parallel() + noTenant := map[string]any{"role": "user"} // validly signed token, no tenant claim + tests := []struct { + name string + filter map[string]Filter + claims map[string]any + }{ + {"_eq", map[string]Filter{"tenant_id": {Eq: new("{{ jwt.tenant }}")}}, noTenant}, + {"_neq, the leak direction", map[string]Filter{"tenant_id": {Neq: new("{{ jwt.tenant }}")}}, noTenant}, + {"_gt", map[string]Filter{"tenant_id": {Gt: new("{{ jwt.tenant }}")}}, noTenant}, + {"_in with surrounding text", map[string]Filter{"tenant_id": {In: new("t-{{ jwt.tenant }}")}}, noTenant}, + { + "object claim in a scalar slot", + map[string]Filter{"tenant_id": {Eq: new("{{ jwt.meta }}")}}, + map[string]any{"meta": map[string]any{"tenant": "acme"}}, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + p := &Policy{Tables: map[string]TablePolicy{ + "t": {"r": {Select: &SelectPermissions{Filter: tt.filter}}}, + }} + perms := Evaluate(p, "r", "t", "select", tt.claims) + + clause, params := perms.Select.WhereSQL(nil) + assert.Equal(t, "1 = 0", clause, "query path: constant-false predicate") + assert.Empty(t, params) + assert.True(t, perms.HasRowFilter(), "failed predicate must keep the stream on the per-subscriber path") + + preds, ok := perms.Predicates() + require.True(t, ok, "a resolved read side answers with its predicates") + require.Len(t, preds, 1, "the failed predicate is present, not dropped") + assert.Equal(t, "tenant_id", preds[0].Column) + assert.Empty(t, preds[0].Values, + "stream path: no values to bind, which matches no row without compiling anything") + }) + } +} + +// TestPredicates_IsTheSameResolutionAsTheWhereClause: the two read surfaces are +// rendered from ONE resolvePredicates call, so the predicates handed to the +// stream carry exactly the values bound into the query's WHERE — in the same +// order. A second resolution, even of the same policy, is what #457 was about. +func TestPredicates_IsTheSameResolutionAsTheWhereClause(t *testing.T) { + t.Parallel() + p := &Policy{Tables: map[string]TablePolicy{ + "t": {"r": {Select: &SelectPermissions{Filter: map[string]Filter{ + "tenant_id": {Eq: new("{{ jwt.tenant }}")}, + }}}}, + }} + perms := Evaluate(p, "r", "t", "select", map[string]any{"tenant": "acme"}) + + clause, params := perms.Select.WhereSQL(nil) + assert.Equal(t, "`tenant_id` = ?", clause) + assert.Equal(t, []any{"acme"}, params) + assert.Equal(t, perms.Select.WhereClause, clause, "WhereSQL(nil) is the WhereClause rendering") + assert.Equal(t, perms.Select.WhereParams, params) + + preds, ok := perms.Predicates() + require.True(t, ok) + assert.Equal(t, []Predicate{{Column: "tenant_id", Op: "=", Values: []string{"acme"}}}, preds) +} + +// TestPredicates_FailsClosedWhereNoRowMayBeAdmitted: a denied grant and an +// INSERT-resolved grant refuse the row question rather than answer "no +// predicates"; a nil receiver (no policy) and a resolved, unfiltered read side +// admit every row. +func TestPredicates_FailsClosedWhereNoRowMayBeAdmitted(t *testing.T) { + t.Parallel() + var none *ResolvedPermissions + preds, ok := none.Predicates() + assert.True(t, ok, "no policy: nothing to filter by") + assert.Empty(t, preds) + + _, ok = (&ResolvedPermissions{Allowed: false}).Predicates() + assert.False(t, ok, "a denied grant admits no row") + + _, ok = (&ResolvedPermissions{Allowed: true, Insert: &ResolvedInsert{}}).Predicates() + assert.False(t, ok, "an unresolved read side admits no row") + + preds, ok = (&ResolvedPermissions{Allowed: true, Select: &ResolvedSelect{}}).Predicates() + assert.True(t, ok) + assert.Empty(t, preds, "a resolved but unfiltered read side admits every row") +} + +// TestWhereSQL_IntegerColumnsBindThroughTheStrictCast: given the column types, +// a claim on an integer column binds as a chsql.IntParam carrying the bare +// integer type (the query builder expands it to chsql.StrictInt), and a claim +// on any other column binds as the plain string it always did. Without types +// (nil) every claim is a plain string. +func TestWhereSQL_IntegerColumnsBindThroughTheStrictCast(t *testing.T) { + t.Parallel() + types := map[string]string{"tenant": "Nullable(UInt64)", "org": "String", "n": "Int128"} + p := &Policy{Tables: map[string]TablePolicy{ + "t": {"r": {Select: &SelectPermissions{Filter: map[string]Filter{ + "tenant": {Eq: new("{{ jwt.tenant }}")}, + "org": {Neq: new("x")}, + "n": {In: new("{{ jwt.ns }}")}, + }}}}, + }} + claims := map[string]any{"tenant": "18446744073709551621", "ns": []any{"1", "-2"}} + perms := Evaluate(p, "r", "t", "select", claims) + require.True(t, perms.Allowed) + + clause, params := perms.Select.WhereSQL(func(c string) string { return types[c] }) + byClause := map[string][]any{} + i := 0 + for part := range strings.SplitSeq(clause, " AND ") { + n := strings.Count(part, "?") + byClause[part] = params[i : i+n] + i += n + } + require.Equal(t, len(params), i, "every ? has exactly one param") + assert.Equal(t, []any{chsql.IntParam{Value: "18446744073709551621", Type: "UInt64"}}, byClause["`tenant` = ?"]) + assert.Equal(t, []any{"x"}, byClause["`org` != ?"]) + assert.Equal(t, []any{chsql.IntParam{Value: "1", Type: "Int128"}, chsql.IntParam{Value: "-2", Type: "Int128"}}, + byClause["`n` IN (?,?)"]) + + _, untyped := perms.Select.WhereSQL(nil) + for _, v := range untyped { + assert.IsType(t, "", v, "no column types: every claim is a plain string") + } +} diff --git a/internal/policy/rowfilter.go b/internal/policy/rowfilter.go index a318e4d4..d4a9b087 100644 --- a/internal/policy/rowfilter.go +++ b/internal/policy/rowfilter.go @@ -144,7 +144,7 @@ func (p *ResolvedPermissions) RowVisible(row map[string]any, cols map[string]Col // matches evaluates one predicate against the row, failing closed (false) whenever // the value is absent or can't be compared as required. -func (pred resolvedPredicate) matches(row map[string]any, spec ColumnSpec) bool { +func (pred Predicate) matches(row map[string]any, spec ColumnSpec) bool { // No values ⇒ matches nothing: an empty/unresolvable "in" set, or a scalar // whose constant was unrenderable — the in-memory twin of the `1 = 0` // predicatesToSQL emits for the same cases. From 6e1fce2764936bc48808323fe8947ac1df51fbf2 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:40:54 -0400 Subject: [PATCH 04/70] feat(discovery): refresh hooks before loaded, and the server zone A SchemaRegistry now runs OnRefresh hooks after each successful Refresh publishes its schemas and before it marks the registry loaded, so a reader that sees Loaded() also sees what the hooks built from the first refresh. Overlapping refreshes run their publish and hooks as one step, in publish order. The server's default zone name is stored with the version and exposed as ServerTimezone. A refresh that finds tables without a CREATE statement (the two system scans are not one snapshot) warns once, naming them. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/discovery/discovery.go | 65 ++++++++++- internal/discovery/discovery_test.go | 166 ++++++++++++++++++++++++++- 2 files changed, 228 insertions(+), 3 deletions(-) diff --git a/internal/discovery/discovery.go b/internal/discovery/discovery.go index 0a7282bf..08a09e23 100644 --- a/internal/discovery/discovery.go +++ b/internal/discovery/discovery.go @@ -223,8 +223,26 @@ type SchemaRegistry struct { // serverVersion is the ClickHouse version string from the last successful // Refresh, guarded by mu alongside tables. serverVersion string + // serverTZ is the server's default time zone name from the same Refresh, + // guarded by mu. ClickHouse reads zone-less timestamps in it, so the type + // layer has to parse in the same zone or it answers about another instant. + serverTZ string + // onRefresh are the hooks a successful Refresh runs with what it + // published; registered before the first Refresh, guarded by mu. + onRefresh []RefreshHook + // publishMu orders a Refresh's publish, its hooks and the loaded flag, so + // two overlapping refreshes (the loop and a manual one) run their hooks in + // the order they published and a hook never sees a snapshot older than + // the one it is replacing. + publishMu sync.Mutex } +// RefreshHook is told what a successful Refresh published: the server's +// version and default time zone name (verbatim from ClickHouse) and every +// discovered table, in no particular order. The schemas are the registry's +// own and must not be modified. +type RefreshHook func(serverVersion, serverTZ string, tables []*TableSchema) + // Source yields a tenant's connection and the database it discovers from, // one snapshot: the database is the one the connection's own pool was // opened for, so the schema discovered always describes the database the @@ -246,9 +264,23 @@ func NewSchemaRegistry(source Source, id tenant.ID, refreshInterval func(tenant. } } +// OnRefresh registers a hook every successful Refresh runs after it publishes +// the new schemas and before it marks the registry loaded, so a reader that +// sees Loaded() also sees every hook's work for the first refresh (the type +// layer binds here). Hooks run synchronously on the refreshing goroutine, in +// registration order, and must be registered before the first Refresh. A +// failed Refresh runs none: the previous schemas stay, and so does whatever +// the hooks built from them. +func (sr *SchemaRegistry) OnRefresh(hook RefreshHook) { + sr.mu.Lock() + defer sr.mu.Unlock() + sr.onRefresh = append(sr.onRefresh, hook) +} + // Refresh rebuilds the in-memory schema cache: it discovers the server's default // time zone and version, queries system.columns, attaches each table's DDL from -// system.tables, and precomputes timestamp column specs. +// system.tables, precomputes timestamp column specs, and then runs the +// OnRefresh hooks before marking the registry loaded. func (sr *SchemaRegistry) Refresh(ctx context.Context) error { tracer := otel.GetTracerProvider().Tracer("wavehouse-discovery") ctx, span := tracer.Start(ctx, "SchemaRegistry.Refresh") @@ -342,17 +374,36 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { return err } + var noDDL []string + published := make([]*TableSchema, 0, len(tables)) for _, ts := range tables { resolveTimestampSpecs(ctx, ts, serverTZ) ts.cacheInsertable() + published = append(published, ts) + if ts.DDL == "" { + noDDL = append(noDDL, ts.Name) + } } + sr.publishMu.Lock() + defer sr.publishMu.Unlock() sr.mu.Lock() sr.tables = tables sr.serverVersion = serverVersion + sr.serverTZ = tzName + hooks := sr.onRefresh sr.mu.Unlock() - sr.loaded.Store(true) slog.InfoContext(ctx, "schema registry refreshed", "tenant", sr.tenant, "tables", len(tables), "server_tz", tzName, "server_version", serverVersion) + if len(noDDL) > 0 { + // The two scans are not one snapshot, so a table can be missing its + // CREATE statement without being missing. One line per refresh, not + // one per table. + slog.WarnContext(ctx, "tables discovered without DDL", "tenant", sr.tenant, "tables", noDDL) + } + for _, hook := range hooks { + hook(serverVersion, tzName, published) + } + sr.loaded.Store(true) return nil } @@ -401,6 +452,16 @@ func (sr *SchemaRegistry) ServerVersion() string { return sr.serverVersion } +// ServerTimezone returns the server's default time zone name captured by the +// last successful Refresh, or "" before the first one. It is the name +// ClickHouse reported, not a resolved location: the type layer hands it +// straight to the parser that reads the rows. +func (sr *SchemaRegistry) ServerTimezone() string { + sr.mu.RLock() + defer sr.mu.RUnlock() + return sr.serverTZ +} + // Get returns the schema for a table, or nil if not found — before the first // refresh as much as for a table the schema lacks, which is the fail-closed // reading the stream hub wants. A handler that answers 404 uses Lookup. diff --git a/internal/discovery/discovery_test.go b/internal/discovery/discovery_test.go index 00bc4073..3456d341 100644 --- a/internal/discovery/discovery_test.go +++ b/internal/discovery/discovery_test.go @@ -438,12 +438,176 @@ func newFakeRegistry(t *testing.T, errs []error) (*SchemaRegistry, *fakeConn) { } // TestRefresh_UnresolvableServerTimezone_NotFatal: an unresolvable server zone -// degrades to pass-through canonicalization (#372), never a failed refresh. +// degrades to pass-through canonicalization (#372), never a failed refresh, +// and the registry still publishes the name verbatim: whoever consumes it +// decides what an unusable zone means. func TestRefresh_UnresolvableServerTimezone_NotFatal(t *testing.T) { t.Parallel() conn := &fakeConn{tz: "Not/AZone"} sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, sr.Refresh(context.Background())) + assert.Equal(t, "Not/AZone", sr.ServerTimezone()) +} + +// TestServerTimezone_EmptyBeforeRefresh: nothing is published until a refresh +// succeeds, so a consumer cannot mistake "not probed yet" for UTC. +func TestServerTimezone_EmptyBeforeRefresh(t *testing.T) { + t.Parallel() + sr, _ := newFakeRegistry(t, nil) + assert.Empty(t, sr.ServerTimezone()) + require.NoError(t, sr.Refresh(context.Background())) + assert.Equal(t, "UTC", sr.ServerTimezone()) +} + +// TestOnRefresh_FiresAfterSwapWithPublishedSchemas: the hook is what binds the +// type layer, so it must see the version, the zone and the same schemas Get() +// now returns — not the ones from before the swap. +func TestOnRefresh_FiresAfterSwapWithPublishedSchemas(t *testing.T) { + t.Parallel() + conn := &fakeConn{ + tz: "Europe/Berlin", + version: "26.6.3.62", + columns: []fakeColumn{{table: "events", name: "id", chType: "UInt64", position: 1}}, + } + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + + var gotVersion, gotTZ string + var gotTables []*TableSchema + calls := 0 + sr.OnRefresh(func(version, tz string, tables []*TableSchema) { + calls++ + gotVersion, gotTZ, gotTables = version, tz, tables + assert.NotNil(t, sr.Get("events"), "hook must run after the swap") + assert.Equal(t, "Europe/Berlin", sr.ServerTimezone(), "the zone is published with the schemas") + }) + + require.NoError(t, sr.Refresh(context.Background())) + assert.Equal(t, 1, calls) + assert.Equal(t, "26.6.3.62", gotVersion) + assert.Equal(t, "Europe/Berlin", gotTZ) + require.Len(t, gotTables, 1) + assert.Equal(t, "events", gotTables[0].Name) + assert.Equal(t, []string{"id"}, gotTables[0].InsertableColumnNames(), "hook sees the memoized schema") +} + +// TestOnRefresh_RunsBeforeLoaded: "loaded" must imply "bound", so on the first +// refresh a Lookup racing the hook still answers ErrNotLoaded (a 503 with +// Retry-After) rather than handing out a schema the type layer has not +// compiled yet. +func TestOnRefresh_RunsBeforeLoaded(t *testing.T) { + t.Parallel() + conn := &fakeConn{columns: []fakeColumn{{table: "events", name: "id", chType: "UInt64", position: 1}}} + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + + var loadedDuringHook []bool + sr.OnRefresh(func(string, string, []*TableSchema) { + loadedDuringHook = append(loadedDuringHook, sr.Loaded()) + _, err := sr.Lookup("events") + if !sr.Loaded() { + assert.ErrorIs(t, err, ErrNotLoaded) + } + }) + + require.NoError(t, sr.Refresh(context.Background())) + assert.True(t, sr.Loaded()) + require.NoError(t, sr.Refresh(context.Background())) + assert.Equal(t, []bool{false, true}, loadedDuringHook, + "first refresh: not loaded until the hook returns; later refreshes stay loaded") +} + +// TestOnRefresh_NotFiredOnFailure: a failed refresh keeps the previous cache, +// so rebinding off a half-read registry would compile the wrong thing. +func TestOnRefresh_NotFiredOnFailure(t *testing.T) { + t.Parallel() + conn := &fakeConn{versionErr: errors.New("server gone")} + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + fired := false + sr.OnRefresh(func(string, string, []*TableSchema) { fired = true }) + require.Error(t, sr.Refresh(context.Background())) + assert.False(t, fired) + assert.False(t, sr.Loaded()) +} + +// TestOnRefresh_HooksRunInRegistrationOrder: more than one consumer may hang +// off a registry, and each sees the same publish. +func TestOnRefresh_HooksRunInRegistrationOrder(t *testing.T) { + t.Parallel() + sr, _ := newFakeRegistry(t, nil) + var order []int + sr.OnRefresh(func(string, string, []*TableSchema) { order = append(order, 1) }) + sr.OnRefresh(func(string, string, []*TableSchema) { order = append(order, 2) }) + require.NoError(t, sr.Refresh(context.Background())) + assert.Equal(t, []int{1, 2}, order) +} + +// TestOnRefresh_OverlappingRefreshesDoNotInterleaveHooks: the manual refresh +// can overlap the auto-refresh loop. Each refresh's publish and hooks run as +// one step, so a hook never runs concurrently with another refresh's hook. +func TestOnRefresh_OverlappingRefreshesDoNotInterleaveHooks(t *testing.T) { + t.Parallel() + conn := &fakeConn{columns: []fakeColumn{{table: "events", name: "id", chType: "UInt64", position: 1}}} + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + var inHook, maxInHook, calls atomic.Int32 + sr.OnRefresh(func(string, string, []*TableSchema) { + n := inHook.Add(1) + for { + m := maxInHook.Load() + if n <= m || maxInHook.CompareAndSwap(m, n) { + break + } + } + time.Sleep(time.Millisecond) + calls.Add(1) + inHook.Add(-1) + }) + + const refreshes = 8 + errs := make(chan error, refreshes) + for range refreshes { + go func() { errs <- sr.Refresh(context.Background()) }() + } + for range refreshes { + require.NoError(t, <-errs) + } + assert.Equal(t, int32(refreshes), calls.Load()) + assert.Equal(t, int32(1), maxInHook.Load(), "hooks of overlapping refreshes must not interleave") +} + +// TestRefresh_WarnsOnceForTablesWithoutDDL: the two scans are not one +// snapshot, so a table can be discovered without its CREATE statement. That +// is one warning per refresh naming every such table, not one per table. +func TestRefresh_WarnsOnceForTablesWithoutDDL(t *testing.T) { + conn := &fakeConn{ + columns: []fakeColumn{ + {table: "a", name: "id", chType: "UInt64", position: 1}, + {table: "b", name: "id", chType: "UInt64", position: 1}, + {table: "c", name: "id", chType: "UInt64", position: 1}, + }, + tables: [][2]string{{"a", "CREATE TABLE test.a (id UInt64) ENGINE = Memory"}}, + } + buf := logtest.Capture(t, slog.LevelWarn) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + require.NoError(t, sr.Refresh(context.Background())) + + out := buf.String() + assert.Equal(t, 1, strings.Count(out, "tables discovered without DDL"), out) + assert.Contains(t, out, `"b"`) + assert.Contains(t, out, `"c"`) + assert.NotContains(t, out, `"a"`) + assert.Equal(t, "CREATE TABLE test.a (id UInt64) ENGINE = Memory", sr.Get("a").DDL) +} + +// TestRefresh_NoWarningWhenEveryTableHasDDL: the warning is for the race, not +// for every refresh. +func TestRefresh_NoWarningWhenEveryTableHasDDL(t *testing.T) { + conn := &fakeConn{ + columns: []fakeColumn{{table: "a", name: "id", chType: "UInt64", position: 1}}, + tables: [][2]string{{"a", "CREATE TABLE test.a (id UInt64) ENGINE = Memory"}}, + } + buf := logtest.Capture(t, slog.LevelWarn) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + require.NoError(t, sr.Refresh(context.Background())) + assert.NotContains(t, buf.String(), "without DDL") } // TestRefresh_RowsIterationError_Fails: rows.Next() returns false on a From 3f274390a0e831c92591139c891c5e02016a1b85 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:42:02 -0400 Subject: [PATCH 05/70] feat(typelayer): tenant-keyed chtypes engine with lazy handle pools Add internal/typelayer, the one package that calls the chtypes SDK (go/v0.4.0, which needs go 1.27). One process-wide Engine holds one lazy registry and every tenant's compiled tables: Bind(tenant, version, zone, tables) compiles a tenant's set, Table and RoleTable hand out read-locked handles, and Forget drops a tenant with its handles closed in the background so a caller holding a reload lock never waits on a request. A missing artifact for a tenant's server line, a server zone that differs from the zone its line was opened with in this process, or a table that does not compile makes that tenant (or that table) Unavailable; every other tenant keeps answering. A table compiles one handle at Bind and grows its pool only when every handle is busy, up to min(GOMAXPROCS, 8); role shapes keep one. A rebind closes the old role projections after releasing the base table, so a request holding a projection can still look the base table up. SkipWithoutArtifact lets any package's tests skip without the artifact, or fail under WAVEHOUSE_TEST_REQUIRE_CHTYPES=1. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- go.mod | 3 +- go.sum | 2 + internal/typelayer/checks_test.go | 517 ++++++++++++++++++++++ internal/typelayer/errors.go | 45 ++ internal/typelayer/filter.go | 299 +++++++++++++ internal/typelayer/filter_test.go | 615 +++++++++++++++++++++++++++ internal/typelayer/ingest.go | 327 ++++++++++++++ internal/typelayer/ingest_test.go | 212 +++++++++ internal/typelayer/pool.go | 211 +++++++++ internal/typelayer/roletable.go | 315 ++++++++++++++ internal/typelayer/roletable_test.go | 301 +++++++++++++ internal/typelayer/tenancy_test.go | 393 +++++++++++++++++ internal/typelayer/testing.go | 94 ++++ internal/typelayer/typelayer.go | 596 ++++++++++++++++++++++++++ internal/typelayer/typelayer_test.go | 429 +++++++++++++++++++ internal/typelayer/zone.go | 52 +++ 16 files changed, 4410 insertions(+), 1 deletion(-) create mode 100644 internal/typelayer/checks_test.go create mode 100644 internal/typelayer/errors.go create mode 100644 internal/typelayer/filter.go create mode 100644 internal/typelayer/filter_test.go create mode 100644 internal/typelayer/ingest.go create mode 100644 internal/typelayer/ingest_test.go create mode 100644 internal/typelayer/pool.go create mode 100644 internal/typelayer/roletable.go create mode 100644 internal/typelayer/roletable_test.go create mode 100644 internal/typelayer/tenancy_test.go create mode 100644 internal/typelayer/testing.go create mode 100644 internal/typelayer/typelayer.go create mode 100644 internal/typelayer/typelayer_test.go create mode 100644 internal/typelayer/zone.go diff --git a/go.mod b/go.mod index 9ff8605e..7baa491a 100644 --- a/go.mod +++ b/go.mod @@ -1,6 +1,6 @@ module github.com/Wave-RF/WaveHouse -go 1.26.6 +go 1.27 tool ( github.com/Zxilly/go-size-analyzer/cmd/gsa @@ -45,6 +45,7 @@ require ( github.com/samber/slog-sampling v1.7.0 github.com/stretchr/testify v1.12.1 github.com/testcontainers/testcontainers-go v0.44.0 + github.com/wave-rf/chtypes/go v0.4.0 go.opentelemetry.io/contrib/bridges/otelslog v0.20.1 go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.71.0 go.opentelemetry.io/contrib/instrumentation/runtime v0.71.0 diff --git a/go.sum b/go.sum index bf6e503b..60a4b185 100644 --- a/go.sum +++ b/go.sum @@ -399,6 +399,8 @@ github.com/tklauser/numcpus v0.12.0 h1:NR85qdvHA9pFse3x3weVZ0r0ST8R6l5RHbZrlRaqo github.com/tklauser/numcpus v0.12.0/go.mod h1:ABHeXzJnr/qqwguhClkZKT1/8VABcYrsyUiUGobwWJg= github.com/vladopajic/go-test-coverage/v2 v2.18.7 h1:Kfpv8jWoC0muCAgk4bR3SvhNfqSQvO9iKZkrhdAZ4dM= github.com/vladopajic/go-test-coverage/v2 v2.18.7/go.mod h1:sTDv3QDUo3Vjo2azG9hyHfMlXH5ugnF1kqDr/4O3w4g= +github.com/wave-rf/chtypes/go v0.4.0 h1:bRNYUzNB+9MoPHmfbE0ebhHipbHPe04jW/7GClI7U4s= +github.com/wave-rf/chtypes/go v0.4.0/go.mod h1:gQE6FgwdtsXvpWlDr79q8XSwDmhiY1mIRa3gj3bOtgo= github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e h1:JVG44RsyaB9T2KIHavMF/ppJZNG9ZpyihvCd0w101no= github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e/go.mod h1:RbqR21r5mrJuqunuUZ/Dhy/avygyECGrLceyNeo4LiM= github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU= diff --git a/internal/typelayer/checks_test.go b/internal/typelayer/checks_test.go new file mode 100644 index 00000000..05042ee5 --- /dev/null +++ b/internal/typelayer/checks_test.go @@ -0,0 +1,517 @@ +package typelayer + +import ( + "fmt" + "strconv" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/wave-rf/chtypes/go/chtypes" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +func checksTable() *discovery.TableSchema { + return &discovery.TableSchema{ + Name: "checks", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt32", Position: 1}, + {Name: "tenant", Type: "String", Position: 2}, + {Name: "kind", Type: "String", Position: 3}, + }, + } +} + +// checksBody is four records whose verdicts under the cases below were +// measured on the 26.6 artifact. +const checksBody = `{"id":1,"tenant":"acme","kind":"a"}` + "\n" + + `{"id":2,"tenant":"acme","kind":"a"}` + "\n" + + `{"id":3,"tenant":"evil","kind":"a"}` + "\n" + + `{"id":4,"tenant":"acme","kind":"z"}` + "\n" + +func checksHandle(t *testing.T) *Table { + t.Helper() + eng := TestEngine(t, checksTable()) + tbl, err := eng.Table(tenant.Default, "checks") + require.NoError(t, err) + t.Cleanup(tbl.Release) + return tbl +} + +// checkReasons is each record's CheckReason, after asserting every record +// parsed and that exactly the admitted ones carry exported bytes. +func checkReasons(t *testing.T, b Batch) []string { + t.Helper() + out := make([]string, len(b.Rows)) + for i, r := range b.Rows { + require.True(t, r.Accepted, "record %d: %s", i, r.Message) + assert.Equal(t, r.CheckReason == "", r.Line != nil, + "record %d: bytes are exported for exactly the admitted records", i) + out[i] = r.CheckReason + } + return out +} + +// TestIngestChecks_MatchesTheMeasuredTable: one AND-joined filter attached to +// the one parse, a verdict per record, and bytes only for the records it +// admits. +func TestIngestChecks_MatchesTheMeasuredTable(t *testing.T) { + tbl := checksHandle(t) + + const f = ReasonFilter + cases := []struct { + name string + preds []Predicate + want []string + }{ + { + "tenant equals", + []Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}}, + []string{"", "", f, ""}, + }, + { + "kind in", + []Predicate{{Column: "kind", Op: "in", Values: []string{"a", "b"}}}, + []string{"", "", "", f}, + }, + { + "both, AND-joined", + []Predicate{ + {Column: "tenant", Op: "=", Values: []string{"acme"}}, + {Column: "kind", Op: "in", Values: []string{"a"}}, + }, + []string{"", "", f, f}, + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(checksBody), tc.preds...) + require.NoError(t, err) + assert.Equal(t, tc.want, checkReasons(t, batch)) + }) + } + + // A record the filter cuts from the middle does not stop the reader: the + // admitted record after it still carries its own bytes. + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(checksBody), + Predicate{Column: "tenant", Op: "=", Values: []string{"acme"}}) + require.NoError(t, err) + assert.Equal(t, `[4, "acme", "z"]`, string(batch.Rows[3].Line)) +} + +// TestIngestChecks_NoPredicatesPassesEveryRow: a role with no check clauses +// pays nothing and hides nothing. +func TestIngestChecks_NoPredicatesPassesEveryRow(t *testing.T) { + tbl := checksHandle(t) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(checksBody)) + require.NoError(t, err) + assert.Equal(t, []string{"", "", "", ""}, checkReasons(t, batch)) + assert.Equal(t, 0, tbl.pool.first().filters.len(), "no filter is compiled without checks") +} + +func TestIngestChecks_EmptyBody(t *testing.T) { + tbl := checksHandle(t) + + batch, err := tbl.Ingest(FormatJSONEachRow, nil, Predicate{Column: "tenant", Op: "=", Values: []string{"acme"}}) + require.NoError(t, err) + assert.Empty(t, batch.Rows) +} + +// TestIngestChecks_FailsClosed: everything that is not a definite true +// withholds, and an unresolvable claim never reaches the compiler. +func TestIngestChecks_FailsClosed(t *testing.T) { + tbl := checksHandle(t) + compiled := func() int { return tbl.pool.first().filters.len() } + const f, e = ReasonFilter, ReasonError + + t.Run("empty values", func(t *testing.T) { + before := compiled() + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(checksBody), + Predicate{Column: "tenant", Op: "=", Values: []string{"acme"}}, + Predicate{Column: "kind", Op: "in", Values: nil}) + require.NoError(t, err) + assert.Equal(t, []string{f, f, f, f}, checkReasons(t, batch)) + assert.Equal(t, before, compiled(), "an unresolvable claim must not reach the compiler") + }) + + t.Run("unknown column", func(t *testing.T) { + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(checksBody), + Predicate{Column: "nosuch", Op: "=", Values: []string{"x"}}) + require.NoError(t, err) + assert.Equal(t, []string{f, f, f, f}, checkReasons(t, batch)) + }) + + t.Run("value the column cannot read", func(t *testing.T) { + // On an integer column the strict cast answers it: no record fits. + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(checksBody), + Predicate{Column: "id", Op: "=", Values: []string{"abc"}}) + require.NoError(t, err) + assert.Equal(t, []string{f, f, f, f}, checkReasons(t, batch), + "an integer claim that does not fit the column is 'the data says no' (403)") + + // On any other column the server's own reader throws. + eng := TestEngine(t, &discovery.TableSchema{Name: "ratios", Columns: []discovery.Column{ + {Name: "id", Type: "UInt32", Position: 1}, + {Name: "ratio", Type: "Float64", Position: 2}, + }}) + ratios, err := eng.Table(tenant.Default, "ratios") + require.NoError(t, err) + t.Cleanup(ratios.Release) + batch, err = ratios.Ingest(FormatJSONEachRow, []byte(`{"id":1,"ratio":0.5}`+"\n"+`{"id":2,"ratio":1}`+"\n"), + Predicate{Column: "ratio", Op: "=", Values: []string{"abc"}}) + require.NoError(t, err) + assert.Equal(t, []string{e, e}, checkReasons(t, batch), + "a thrown predicate is 'we could not tell' (422), not 'the data says no' (403)") + assert.Contains(t, batch.Rows[0].Message, "abc", "the predicate's own error rides along for the log") + }) + + t.Run("a filter closed between lookup and use", func(t *testing.T) { + preds := []Predicate{{Column: "tenant", Op: "=", Values: []string{"closed"}}} + expr, params, ok := tbl.render(preds) + require.True(t, ok) + for _, s := range tbl.pool.list() { // whichever slot the request lands on + f := tbl.filterOn(s, expr, params) + require.NotNil(t, f) + f.Close() // what an eviction racing this request does + } + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(checksBody), preds...) + require.NoError(t, err, "a closed filter fails the checks, not the request") + assert.Equal(t, []string{ReasonDecline, ReasonDecline, ReasonDecline, ReasonDecline}, checkReasons(t, batch)) + }) +} + +// TestIngestChecks_IntegerClaimThatDoesNotFitIsRefused: the insert check +// compares an integer claim through the strict cast, so a claim the column +// cannot hold refuses every record — including one whose value the plain +// String binding would have wrapped the claim onto (2^64+5 → 5), and one filled +// from the injected DEFAULT, which the compiler may itself wrap. A claim that +// fits admits exactly as before. +func TestIngestChecks_IntegerClaimThatDoesNotFitIsRefused(t *testing.T) { + eng := TestEngine(t, ordersTable()) + const over = "18446744073709551621" // 2^64+5 + body := []byte(`{"id":1}` + "\n" + `{"id":2,"amount":5}` + "\n" + `{"id":3,"amount":6}` + "\n") + + t.Run("a claim past the column's range", func(t *testing.T) { + tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"amount": over}}) + batch, err := tbl.Ingest(FormatJSONEachRow, body, Predicate{Column: "amount", Op: "=", Values: []string{over}}) + require.NoError(t, err) + assert.Equal(t, []string{ReasonFilter, ReasonFilter, ReasonFilter}, checkReasons(t, batch)) + }) + + t.Run("a claim that fits", func(t *testing.T) { + tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"amount": "5"}}) + batch, err := tbl.Ingest(FormatJSONEachRow, body, Predicate{Column: "amount", Op: "=", Values: []string{"5"}}) + require.NoError(t, err) + assert.Equal(t, []string{"", "", ReasonFilter}, checkReasons(t, batch)) + assert.Equal(t, `[1, "", "", 5]`, string(batch.Rows[0].Line)) + }) + + t.Run("an _in set keeps only the elements that fit", func(t *testing.T) { + tbl := checksHandleFor(t, eng, "orders") + in := []byte(`{"id":1,"amount":5}` + "\n" + `{"id":2,"amount":0}` + "\n" + + `{"id":3,"amount":"18446744073709551615"}` + "\n" + `{"id":4,"amount":7}` + "\n") + batch, err := tbl.Ingest(FormatJSONEachRow, in, Predicate{ + Column: "amount", Op: "in", + Values: []string{over, "007", "0", "115792089237316195423570985008687907853269984665640564039457584007913129639941", "+7"}, + }) + require.NoError(t, err) + assert.Equal(t, []string{ReasonFilter, "", ReasonFilter, ReasonFilter}, checkReasons(t, batch)) + }) +} + +func checksHandleFor(t *testing.T, eng *Engine, table string) *Table { + t.Helper() + tbl, err := eng.Table(tenant.Default, table) + require.NoError(t, err) + t.Cleanup(tbl.Release) + return tbl +} + +// TestIngestChecks_ParseOutcomeDecidesFirst pins a measured trap: under the +// compile profile's allow_errors_ratio a record that does not parse is +// skipped, and chtypes answers it 'd' beside that outcome — with the verdict's +// own code and message EMPTY on 26.6. It must report its parse error (a 400 +// with code 27), never a check decline (a 422), and never shift a neighbour +// onto its answer. +func TestIngestChecks_ParseOutcomeDecidesFirst(t *testing.T) { + tbl := checksHandle(t) + body := []byte(strings.Join([]string{ + `{"id":1,"tenant":"acme","kind":"a"}`, + `{"id":"not-a-number","tenant":"acme","kind":"a"}`, + `{"id":3,"tenant":"evil","kind":"a"}`, + `{"id":4,"tenant":"acme","kind":"a"}`, + }, "\n") + "\n") + preds := []Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}} + + // The trap itself, at the SDK: a skipped row's verdict is 'd' with no + // verdict code, and its error is only in ErrCode/ErrMsg. + s := tbl.pool.first() + expr, params, ok := tbl.render(preds) + require.True(t, ok) + res, err := s.schema.RowsExportWith(FormatJSONEachRow, body, InsertSettings(), chtypes.JSONCompactEachRow, + chtypes.WithRowFilter(tbl.filterOn(s, expr, params))) + require.NoError(t, err) + require.Len(t, res.Rows, 4) + require.Equal(t, chtypes.Skipped, res.Rows[1].Outcome) + require.NotNil(t, res.Rows[1].Verdict) + assert.Equal(t, chtypes.VerdictDecline, *res.Rows[1].Verdict) + assert.Equal(t, 27, res.Rows[1].ErrCode) + assert.Equal(t, 2, res.RowsPassed) + assert.Equal(t, 1, res.RowsCut, "a skipped row is 'd' but not counted as cut") + + batch, err := tbl.Ingest(FormatJSONEachRow, body, preds...) + require.NoError(t, err) + require.Len(t, batch.Rows, 4) + assert.True(t, batch.Rows[0].Accepted) + assert.Empty(t, batch.Rows[0].CheckReason) + assert.Equal(t, `[1, "acme", "a"]`, string(batch.Rows[0].Line)) + + assert.False(t, batch.Rows[1].Accepted) + assert.False(t, batch.Rows[1].Declined, "a parse refusal is a verdict about the data") + assert.Equal(t, 27, batch.Rows[1].Code) + assert.Empty(t, batch.Rows[1].CheckReason) + + assert.Equal(t, ReasonFilter, batch.Rows[2].CheckReason) + assert.Nil(t, batch.Rows[2].Line) + assert.Equal(t, `[4, "acme", "a"]`, string(batch.Rows[3].Line)) +} + +// TestIngestChecks_RejectedBatchExportsNothing pins the other measured trap: +// RowsPassed counts the admitted rows of a batch whose own outcome is +// rejected, and such a batch exports no bytes. It is reachable here — an +// NDJSON-declared single-line array is not reframed (only the JSON family's +// arrays are), and one bad element rejects it whole. Every record must be +// declined; none may be published on the strength of RowsPassed. +func TestIngestChecks_RejectedBatchExportsNothing(t *testing.T) { + tbl := checksHandle(t) + body := []byte(`[{"id":1,"tenant":"acme","kind":"a"},{"id":"x","tenant":"acme","kind":"a"},{"id":3,"tenant":"acme","kind":"a"}]`) + preds := []Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}} + + s := tbl.pool.first() + expr, params, ok := tbl.render(preds) + require.True(t, ok) + res, err := s.schema.RowsExportWith(FormatJSONEachRow, body, InsertSettings(), chtypes.JSONCompactEachRow, + chtypes.WithRowFilter(tbl.filterOn(s, expr, params))) + require.NoError(t, err) + require.Equal(t, chtypes.Rejected, res.Outcome) + require.Positive(t, res.RowsPassed, "the trap: an admitted row is counted in a rejected batch") + require.Empty(t, res.Payload) + + batch, err := tbl.Ingest(FormatJSONEachRow, body, preds...) + require.NoError(t, err) + require.NotEmpty(t, batch.Rows) + for i, r := range batch.Rows { + assert.False(t, r.Accepted, "record %d", i) + assert.True(t, r.Declined, "record %d", i) + assert.Nil(t, r.Line, "record %d", i) + } +} + +// TestIngestChecks_OnTheWithNamesFormats: the header is not a record, so the +// verdicts line up with the data lines whatever order the header names the +// columns in. +func TestIngestChecks_OnTheWithNamesFormats(t *testing.T) { + tbl := checksHandle(t) + + batch, err := tbl.Ingest(FormatCSVWithNames, []byte("tenant,id,kind\nacme,1,a\nevil,2,a\nacme,x,a\nacme,4,a\n"), + Predicate{Column: "tenant", Op: "=", Values: []string{"acme"}}) + require.NoError(t, err) + require.Len(t, batch.Rows, 4) + assert.Equal(t, `[1, "acme", "a"]`, string(batch.Rows[0].Line)) + assert.Equal(t, ReasonFilter, batch.Rows[1].CheckReason) + assert.Equal(t, 27, batch.Rows[2].Code) + assert.Equal(t, `[4, "acme", "a"]`, string(batch.Rows[3].Line)) +} + +func TestIngest_UnavailableTable(t *testing.T) { + eng := TestEngine(t, checksTable()) + eng.Bind(tenant.Default, TestServerVersion, "UTC", nil) + + _, err := eng.Table(tenant.Default, "checks") + require.Error(t, err) + assert.True(t, IsUnavailable(err)) +} + +func TestIngest_RefusesAFormatItCannotParse(t *testing.T) { + tbl := checksHandle(t) + + for _, f := range []Format{chtypes.JSONCompactEachRow, chtypes.RowBinary, chtypes.Native, Format(99)} { + _, err := tbl.Ingest(f, []byte(`{"id":1}`+"\n")) + require.Error(t, err, "format %d", int(f)) + assert.False(t, IsUnavailable(err), "a bad format is a programming error, not an outage") + assert.Contains(t, err.Error(), "FormatJSONEachRow") + } +} + +// TestIngest_PositionalFormats: CSV and TSV are declaration-order positional; +// header detection is ClickHouse's default and StrictPositional switches it off. +func TestIngest_PositionalFormats(t *testing.T) { + tbl := checksHandle(t) + + csv, err := tbl.Ingest(FormatCSV, []byte("1,acme,a\n2,evil,z\n")) + require.NoError(t, err) + require.Len(t, csv.Rows, 2) + require.True(t, csv.Rows[0].Accepted, csv.Rows[0].Message) + assert.Equal(t, `[1, "acme", "a"]`, string(csv.Rows[0].Line)) + + tsv, err := tbl.Ingest(FormatTSV, []byte("1\tacme\ta\n")) + require.NoError(t, err) + require.Len(t, tsv.Rows, 1) + require.True(t, tsv.Rows[0].Accepted, tsv.Rows[0].Message) + assert.Equal(t, `[1, "acme", "a"]`, string(tsv.Rows[0].Line)) + + // With StrictPositional a header line is one failed record with ClickHouse's + // own code 27, even one that names every column; by default ClickHouse + // consumes it as a header (input_format_*_detect_header is on by default), + // so the same body is one data row. + for name, body := range map[string][]byte{ + "csv": []byte("id,tenant,kind\n1,acme,a\n"), + "tsv": []byte("id\ttenant\tkind\n1\tacme\ta\n"), + } { + format := FormatCSV + if name == "tsv" { + format = FormatTSV + } + strict, err := tbl.IngestWith(format, IngestOptions{StrictPositional: true}, body) + require.NoError(t, err) + require.Len(t, strict.Rows, 2, name) + assert.False(t, strict.Rows[0].Accepted, name) + assert.Equal(t, 27, strict.Rows[0].Code, name) + assert.True(t, strict.Rows[1].Accepted, name) + + detected, err := tbl.Ingest(format, body) + require.NoError(t, err) + require.Len(t, detected.Rows, 1, name) + assert.True(t, detected.Rows[0].Accepted, name) + } +} + +// TestIngest_WithNamesFormats pins the header formats as measured on the 26.6 +// artifact: the header line is not a record (Rows index the data lines), it +// names the columns in any order, a column it omits takes its DEFAULT, and a +// name the schema lacks or a repeated name refuses the body whole with +// ClickHouse's own 117. +func TestIngest_WithNamesFormats(t *testing.T) { + tbl := checksHandle(t) + + for _, tc := range []struct { + format Format + sep string + }{{FormatCSVWithNames, ","}, {FormatTSVWithNames, "\t"}} { + body := func(lines ...string) []byte { + for i, l := range lines { + lines[i] = strings.ReplaceAll(l, ",", tc.sep) + } + return []byte(strings.Join(lines, "\n") + "\n") + } + name := fmt.Sprintf("format %d", int(tc.format)) + + b, err := tbl.Ingest(tc.format, body("kind,id,tenant", "a,1,acme", "z,x,acme", "b,3,evil")) + require.NoError(t, err, name) + require.Len(t, b.Rows, 3, "%s: the header is not a record", name) + assert.Equal(t, `[1, "acme", "a"]`, string(b.Rows[0].Line), name) + assert.Equal(t, 27, b.Rows[1].Code, name) + assert.Equal(t, `[3, "evil", "b"]`, string(b.Rows[2].Line), name) + + b, err = tbl.Ingest(tc.format, body("id,tenant", "1,acme")) + require.NoError(t, err, name) + require.Len(t, b.Rows, 1, name) + assert.Equal(t, `[1, "acme", ""]`, string(b.Rows[0].Line), "%s: an omitted column takes its DEFAULT", name) + + b, err = tbl.Ingest(tc.format, body("id,tenant,kind")) + require.NoError(t, err, name) + assert.Empty(t, b.Rows, "%s: a header alone is zero records", name) + assert.Nil(t, b.Refused, name) + + for _, header := range []string{"id,tenant,kind,extra", "id,id,kind"} { + b, err = tbl.Ingest(tc.format, body(header, "1,acme,a,z")) + require.NoError(t, err, name) + require.NotNil(t, b.Refused, "%s: %s", name, header) + assert.Equal(t, 117, b.Refused.Code, "%s: %s", name, header) + assert.NotEmpty(t, b.Refused.Message) + assert.Empty(t, b.Rows) + } + } + + // Header names match case-insensitively from 26.5, as the server does. + b, err := tbl.Ingest(FormatCSVWithNames, []byte("ID,Tenant,KIND\n1,acme,a\n")) + require.NoError(t, err) + require.Len(t, b.Rows, 1) + assert.True(t, b.Rows[0].Accepted, b.Rows[0].Message) +} + +// BenchmarkIngest_HandlePool is the standing evidence behind maxPoolSize (re-run it on deployment hardware). +// Three arms, identical work, only the concurrency and the handle differ: +// +// - serial: one goroutine, one handle — the cost of a call with no contention +// - parallel-shared: GOMAXPROCS goroutines, ONE handle +// - parallel-pooled: GOMAXPROCS goroutines, the whole pool +// +// Read it this way: if parallel-shared's ns/op is not meaningfully better than +// serial's, RowsExport is not parallelizing at all and a pool of handles +// cannot help — which is what darwin shows, while Linux scales (see maxPoolSize). +// Only when parallel-shared is ~GOMAXPROCS× worse than serial does a pool have +// anything to win, and parallel-pooled is then the size of the win. +func BenchmarkIngest_HandlePool(b *testing.B) { + eng := TestEngine(b, checksTable()) + tbl, err := eng.Table(tenant.Default, "checks") + require.NoError(b, err) + defer tbl.Release() + + var body strings.Builder + for i := range 500 { + body.WriteString(`{"id":` + strconv.Itoa(i) + `,"tenant":"acme","kind":"a"}` + "\n") + } + raw := []byte(body.String()) + + // Identical work on both arms — only the handle choice differs. + export := func(b *testing.B, pick func() *schemaSlot) { + b.Helper() + b.ResetTimer() + b.RunParallel(func(pb *testing.PB) { + for pb.Next() { + if _, err := pick().schema.RowsExport( + FormatJSONEachRow, raw, InsertSettings(), chtypes.JSONCompactEachRow); err != nil { + b.Fatal(err) + } + } + }) + } + + b.Run("serial", func(b *testing.B) { + s := tbl.pool.first() + b.ResetTimer() + for range b.N { + if _, err := s.schema.RowsExport( + FormatJSONEachRow, raw, InsertSettings(), chtypes.JSONCompactEachRow); err != nil { + b.Fatal(err) + } + } + }) + b.Run("parallel-shared", func(b *testing.B) { + export(b, func() *schemaSlot { return tbl.pool.first() }) + }) + b.Run("parallel-pooled", func(b *testing.B) { + // Every slot the pool may grow to, so the arm measures the whole pool + // rather than whatever the lazy growth has reached so far. + p := tbl.pool + held := make([]*schemaSlot, 0, poolSize()) + for range poolSize() { + held = append(held, p.acquire()) + } + for _, s := range held { + p.release(s) + } + export(b, func() *schemaSlot { + s := p.acquire() + p.release(s) + return s + }) + }) +} diff --git a/internal/typelayer/errors.go b/internal/typelayer/errors.go new file mode 100644 index 00000000..14867bc9 --- /dev/null +++ b/internal/typelayer/errors.go @@ -0,0 +1,45 @@ +package typelayer + +import ( + "errors" + "fmt" + + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// Unavailable reports that no compiled schema can answer for a table right +// now. It is never a verdict about data: callers map it to HTTP 503 on ingest +// and to "withhold" on the stream, so it must stay distinguishable from a +// ClickHouse rejection. +// +// It is always about one tenant: a missing artifact for the tenant's server +// line, a server zone this process cannot adopt, or a table that did not +// compile leaves every other tenant answering. +type Unavailable struct { + Tenant tenant.ID + // Table is "" when the cause covers the tenant's every table. + Table string + // Cause is the operator-facing reason, already carrying the SDK's own + // wording where there is one (an artifact-missing error lists the + // directories it searched, which is the whole diagnostic). + Cause string +} + +func (e *Unavailable) Error() string { + if e.Table == "" { + return fmt.Sprintf("chtypes unavailable for tenant %q: %s", e.Tenant, e.Cause) + } + return fmt.Sprintf("chtypes unavailable for tenant %q table %q: %s", e.Tenant, e.Table, e.Cause) +} + +// IsUnavailable reports whether err is an *Unavailable anywhere in its chain. +func IsUnavailable(err error) bool { + var u *Unavailable + return errors.As(err, &u) +} + +// ErrColumnsDrift is returned by ParseRow when the envelope's column list is +// not the one the current compiled handle exports. A positional row is only +// interpretable against the generation that produced it, so a mismatch means +// the event predates a schema change and must be withheld rather than guessed. +var ErrColumnsDrift = errors.New("row columns do not match the table's wire columns") diff --git a/internal/typelayer/filter.go b/internal/typelayer/filter.go new file mode 100644 index 00000000..14cce201 --- /dev/null +++ b/internal/typelayer/filter.go @@ -0,0 +1,299 @@ +package typelayer + +import ( + "container/list" + "encoding/json" + "fmt" + "log/slog" + "slices" + "strings" + "sync" + + "github.com/wave-rf/chtypes/go/chtypes" + + "github.com/Wave-RF/WaveHouse/internal/chsql" + "github.com/Wave-RF/WaveHouse/internal/policy" +) + +// filterCacheSize bounds the compiled filters held per table. A filter handle +// is identified by (expression, bound values), and the values come from tenant +// claims, so an unbounded cache is a memory/CPU denial of service. The budget +// is split across the handle pool's limit (see newPool), so this is the +// table's total, not each slot's. +const filterCacheSize = 4096 + +// Predicate is one resolved row-filter clause. Values are the canonical strings +// policy already resolved for the SQL path, so the stream and the query answer +// off one resolution; len(Values)==0 means "matches nothing" and never widens. +type Predicate = policy.Predicate + +// Reason names why a row was not visible, for the withheld-reason metric. +const ( + ReasonFilter = "filter" // the predicate answered false + ReasonError = "error" // the predicate threw on this row's values + ReasonDecline = "decline" // chtypes would not answer +) + +// Row is one parsed event, reusable across every subscriber's filter. Parsing +// is the expensive half, so the hub parses once per event and evaluates K +// filters against the result. +// +// The Table it came from must stay held (not Released) until Close, and Close +// is required: the Row keeps its handle busy until then. +type Row struct { + table *Table + pool *pool + slot *schemaSlot + block *chtypes.LoadedBlock +} + +// ParseRow parses one JSONCompactEachRow line, with or without its trailing +// newline. columns must be the generation's wire columns exactly: a positional +// row is uninterpretable against any other order, so a mismatch is +// ErrColumnsDrift rather than a guess. +func (t *Table) ParseRow(columns []string, row []byte) (*Row, error) { + if t.pool == nil { + return nil, &Unavailable{Tenant: t.tenant, Table: t.Name, Cause: t.cause} + } + if !slices.Equal(columns, t.WireColumns) { + return nil, fmt.Errorf("%w: event carries %v, generation %d exports %v", + ErrColumnsDrift, columns, t.Generation, t.WireColumns) + } + body := row + if n := len(body); n == 0 || body[n-1] != '\n' { + body = append(append(make([]byte, 0, n+1), body...), '\n') + } + // The block and every filter evaluated against it stay on ONE handle: a + // cross-handle Eval takes both handles' locks and hands back exactly the + // serialization the pool exists to avoid. + p := t.pool + s := p.acquire() + block, err := s.schema.ParseBlock(chtypes.JSONCompactEachRow, body, InsertSettings()) + if err != nil { + p.release(s) + return nil, err + } + return &Row{table: t, pool: p, slot: s, block: block}, nil +} + +// Close frees the parsed block and gives its handle back. Required: the C +// layer does not refcount. A second Close is a no-op. +func (r *Row) Close() { + if r.block != nil { + r.block.Close() + r.block = nil + r.pool.release(r.slot) + } +} + +// Visible reports whether the row satisfies every predicate. Only a definite +// true is visible — false, a predicate that threw, and a decline all withhold. +func (r *Row) Visible(preds []Predicate) bool { + ok, _ := r.VisibleWithReason(preds) + return ok +} + +// VisibleWithReason is Visible plus the label the withheld-reason metric wants. +// The reason is "" when the row is visible. +func (r *Row) VisibleWithReason(preds []Predicate) (bool, string) { + if len(preds) == 0 { + return true, "" + } + if r.block == nil { + return false, ReasonDecline + } + expr, params, ok := r.table.render(preds) + if !ok { + // An unresolvable predicate matches nothing, exactly as the SQL path's + // `1 = 0` does — the two surfaces must not disagree (#457). + return false, ReasonFilter + } + + filter := r.table.filterOn(r.slot, expr, params) + if filter == nil { + return false, ReasonDecline + } + res, err := filter.Eval(r.block) + if err != nil || res.Outcome != chtypes.FilterOK || len(res.Verdicts) == 0 { + return false, ReasonDecline + } + return verdictBool(res.Verdicts[0]) +} + +// verdictBool maps one chtypes verdict onto (visible, reason) — for a stored +// row's visibility and for an ingested record's insert check alike. Only +// VerdictTrue is true, and the zero value is Decline, so an answer nobody set +// withholds. +func verdictBool(v chtypes.Verdict) (bool, string) { + switch v { + case chtypes.VerdictTrue: + return true, "" + case chtypes.VerdictFalse: + return false, ReasonFilter + case chtypes.VerdictError: + return false, ReasonError + case chtypes.VerdictDecline: + return false, ReasonDecline + default: + return false, ReasonDecline + } +} + +// render builds the AND-joined expression and the parameter map. +// +// Every value binds as a {pN:String} parameter, whatever the column's declared +// type — the same binding the SQL path uses, so both surfaces read a claim with +// ClickHouse's own comparison-time coercion (a typed parameter disagreed with +// the server on UInt8, Int64 and Float32 columns; the String binding matched +// it on every column family measured): a spelling the column cannot read +// (`abc` on a Float32 column) is the server's own code 53 at evaluation, +// which withholds the row. A bare String binding still wraps an +// integer value at or past 2^64 before comparing (and a 128/256-bit column at +// its own width), so on an integer column the parameter is compared through +// chsql.StrictInt instead: a claim that is not the canonical spelling of a +// value the column can hold is NULL and matches nothing on any operator, and +// an in-range claim answers exactly as the plain binding does. The query path +// renders the same expression (policy.ResolvedSelect.WhereSQL). +// +// Every value is encoded with chsql.EscapeStringParam, the SAME encoding the +// SQL path uses for its `{p:String}` parameters. The artifact reads a filter +// parameter with ClickHouse's escaped-text reader, exactly as the server reads +// one off the HTTP interface: measured on the 26.6 artifact, a stored `a\b` +// compared false against the raw value (the `\b` read as a backspace), and a +// stored tab, newline or trailing backslash would not compile at all (a +// decline — every row withheld). With the encoding applied all of them compare +// equal, on `=` and on `in` alike. Before this, a claim carrying any of those +// bytes silently withheld rows the SQL path returned. +// +// Identifiers are the library's own QuoteIdentifier spelling (see +// declaredColumns). Values are never interpolated, so a hostile claim is inert +// by construction. Reports false when a predicate cannot be expressed, which +// fails closed without compiling anything. +func (t *Table) render(preds []Predicate) (string, map[string]string, bool) { + var b strings.Builder + params := make(map[string]string, len(preds)) + n := 0 + bind := func(col filterColumn, v string) { + name := fmt.Sprintf("p%d", n) + n++ + params[name] = chsql.EscapeStringParam(v) + if col.intType != "" { + b.WriteString(chsql.StrictInt(name, col.intType)) + return + } + fmt.Fprintf(&b, "{%s:String}", name) + } + for i, p := range preds { + col, known := t.cols[p.Column] + if !known || len(p.Values) == 0 { + return "", nil, false + } + if i > 0 { + b.WriteString(" AND ") + } + b.WriteString(col.ident) + switch p.Op { + case "=", "!=", ">", "<": + if len(p.Values) != 1 { + return "", nil, false + } + fmt.Fprintf(&b, " %s ", p.Op) + bind(col, p.Values[0]) + case "in": + b.WriteString(" IN (") + for j, v := range p.Values { + if j > 0 { + b.WriteString(", ") + } + bind(col, v) + } + b.WriteByte(')') + default: + return "", nil, false + } + } + return b.String(), params, true +} + +// filterOn returns the compiled handle for this expression and parameter set +// on ONE slot, compiling it at most once per (slot, generation). A compile +// failure is cached as a negative entry so a broken policy costs one compile +// and one log line, not one per event. +func (t *Table) filterOn(s *schemaSlot, expr string, params map[string]string) *chtypes.LoadedFilter { + key := cacheKey(t.Generation, expr, params) + c := s.filters + + c.mu.Lock() + defer c.mu.Unlock() + if el, hit := c.index[key]; hit { + c.order.MoveToFront(el) + return el.Value.(*filterEntry).filter + } + + f, err := s.schema.CompileFilter(expr, chtypes.WithFilterParams(params)) + if err != nil { + f = nil + slog.Error("row filter will not compile; withholding every row for it", + "tenant", t.tenant, "table", t.Name, "generation", t.Generation, "expr", expr, "error", err) + } + el := c.order.PushFront(&filterEntry{key: key, filter: f}) + c.index[key] = el + if c.order.Len() > c.cap { + c.evictOldestLocked() + } + return f +} + +// cacheKey identifies a compiled handle. Values are baked into the handle at +// compile time, so they belong in the key alongside the generation that owns +// the schema. +func cacheKey(generation uint64, expr string, params map[string]string) string { + // The parameter names are positional (p0, p1 …) and generated from the same + // expression, so a JSON object over them is stable without sorting. + enc, _ := json.Marshal(params) + return fmt.Sprintf("%d\x00%s\x00%s", generation, expr, enc) +} + +type filterEntry struct { + key string + filter *chtypes.LoadedFilter // nil: this expression does not compile +} + +type filterCache struct { + mu sync.Mutex + cap int + order *list.List // front = most recently used + index map[string]*list.Element +} + +func newFilterCache(capacity int) *filterCache { + return &filterCache{cap: capacity, order: list.New(), index: make(map[string]*list.Element)} +} + +func (c *filterCache) evictOldestLocked() { + el := c.order.Back() + if el == nil { + return + } + c.order.Remove(el) + e := el.Value.(*filterEntry) + delete(c.index, e.key) + if e.filter != nil { + e.filter.Close() + } +} + +// closeAll drops every handle. Called before the schema is closed so no freed +// pointer survives in the index; the schema would close them anyway, but not +// the map holding them. +func (c *filterCache) closeAll() { + c.mu.Lock() + defer c.mu.Unlock() + for el := c.order.Front(); el != nil; el = el.Next() { + if f := el.Value.(*filterEntry).filter; f != nil { + f.Close() + } + } + c.order.Init() + c.index = make(map[string]*list.Element) +} diff --git a/internal/typelayer/filter_test.go b/internal/typelayer/filter_test.go new file mode 100644 index 00000000..fe1277ec --- /dev/null +++ b/internal/typelayer/filter_test.go @@ -0,0 +1,615 @@ +package typelayer + +import ( + "encoding/json" + "fmt" + "math/big" + "strconv" + "strings" + "sync" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// rowsTable carries one column of each family the row filter has to compare, +// including the three (UInt8, Int64, Float32) where a typed parameter used to +// disagree with the server and a String parameter does not. +func rowsTable() *discovery.TableSchema { + return &discovery.TableSchema{ + Name: "rows", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt8", Position: 1}, + {Name: "tenant", Type: "String", Position: 2}, + {Name: "ts", Type: "DateTime", Position: 3}, + {Name: "amt", Type: "Decimal(10,2)", Position: 4}, + {Name: "big", Type: "Int64", Position: 5}, + {Name: "ratio", Type: "Float32", Position: 6}, + {Name: "tags", Type: "Array(String)", Position: 7}, + }, + } +} + +const sampleRow = `[7, "acme", "2026-01-15 10:30:00", "12.50", -5, 0.1, ["a"]]` + +func parsedRow(t *testing.T) (*Table, *Row) { + t.Helper() + eng := TestEngine(t, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + t.Cleanup(tbl.Release) + + row, err := tbl.ParseRow(tbl.WireColumns, []byte(sampleRow)) + require.NoError(t, err) + t.Cleanup(row.Close) + return tbl, row +} + +// TestVisible_StringBindingAcrossColumnFamilies is the pin for the String +// binding: every value binds as {pN:String} and the answer still matches +// the SQL path on every operator and every column family. The row is +// id=7, tenant="acme", ts=2026-01-15 10:30:00, amt=12.50, big=-5, ratio=0.1. +func TestVisible_StringBindingAcrossColumnFamilies(t *testing.T) { + _, row := parsedRow(t) + + cases := []struct { + name string + pred Predicate + want bool + }{ + {"string equal", Predicate{Column: "tenant", Op: "=", Values: []string{"acme"}}, true}, + {"string equal miss", Predicate{Column: "tenant", Op: "=", Values: []string{"beta"}}, false}, + {"string not equal", Predicate{Column: "tenant", Op: "!=", Values: []string{"beta"}}, true}, + {"string not equal self", Predicate{Column: "tenant", Op: "!=", Values: []string{"acme"}}, false}, + {"string greater", Predicate{Column: "tenant", Op: ">", Values: []string{"aaa"}}, true}, + {"string less", Predicate{Column: "tenant", Op: "<", Values: []string{"aaa"}}, false}, + {"string in", Predicate{Column: "tenant", Op: "in", Values: []string{"acme", "beta"}}, true}, + {"string in miss", Predicate{Column: "tenant", Op: "in", Values: []string{"beta", "gamma"}}, false}, + + {"uint equal", Predicate{Column: "id", Op: "=", Values: []string{"7"}}, true}, + {"uint not equal", Predicate{Column: "id", Op: "!=", Values: []string{"8"}}, true}, + {"uint greater", Predicate{Column: "id", Op: ">", Values: []string{"3"}}, true}, + {"uint less", Predicate{Column: "id", Op: "<", Values: []string{"3"}}, false}, + {"uint in", Predicate{Column: "id", Op: "in", Values: []string{"1", "7"}}, true}, + + {"int64 equal", Predicate{Column: "big", Op: "=", Values: []string{"-5"}}, true}, + {"int64 not equal", Predicate{Column: "big", Op: "!=", Values: []string{"-5"}}, false}, + {"int64 greater", Predicate{Column: "big", Op: ">", Values: []string{"-6"}}, true}, + {"int64 less", Predicate{Column: "big", Op: "<", Values: []string{"-6"}}, false}, + {"int64 in", Predicate{Column: "big", Op: "in", Values: []string{"-5", "0"}}, true}, + + // The #381 storage-narrowing case: a Float32 column's 0.1 is + // 0.100000001490116…, so a constant widened to Float64 would NOT equal + // it and the stream would then admit `!= '0.1'` on a row /v1/query + // hides. A String parameter is read in the column's own domain. + {"float32 equal", Predicate{Column: "ratio", Op: "=", Values: []string{"0.1"}}, true}, + {"float32 not equal", Predicate{Column: "ratio", Op: "!=", Values: []string{"0.1"}}, false}, + {"float32 greater", Predicate{Column: "ratio", Op: ">", Values: []string{"0.05"}}, true}, + {"float32 less", Predicate{Column: "ratio", Op: "<", Values: []string{"0.05"}}, false}, + {"float32 in", Predicate{Column: "ratio", Op: "in", Values: []string{"0.1", "2"}}, true}, + + {"decimal equal", Predicate{Column: "amt", Op: "=", Values: []string{"12.50"}}, true}, + {"decimal equal shorter spelling", Predicate{Column: "amt", Op: "=", Values: []string{"12.5"}}, true}, + {"decimal not equal", Predicate{Column: "amt", Op: "!=", Values: []string{"12.49"}}, true}, + {"decimal greater", Predicate{Column: "amt", Op: ">", Values: []string{"12.49"}}, true}, + {"decimal less", Predicate{Column: "amt", Op: "<", Values: []string{"12.49"}}, false}, + {"decimal in", Predicate{Column: "amt", Op: "in", Values: []string{"12.50", "1"}}, true}, + + {"datetime equal", Predicate{Column: "ts", Op: "=", Values: []string{"2026-01-15 10:30:00"}}, true}, + {"datetime not equal", Predicate{Column: "ts", Op: "!=", Values: []string{"2026-01-15 10:30:00"}}, false}, + {"datetime greater", Predicate{Column: "ts", Op: ">", Values: []string{"2026-01-01 00:00:00"}}, true}, + {"datetime less", Predicate{Column: "ts", Op: "<", Values: []string{"2026-01-01 00:00:00"}}, false}, + {"datetime in", Predicate{Column: "ts", Op: "in", Values: []string{"2026-01-15 10:30:00"}}, true}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + assert.Equal(t, tc.want, row.Visible([]Predicate{tc.pred})) + }) + } +} + +// TestVisible_HostileSpellingsMatchNothing: on an integer column a claim is +// compared through the strict cast, so a value outside the column's domain, or +// a spelling that is not its canonical form, matches nothing on EVERY operator +// — `!=` included — and is answered false rather than thrown. On a +// non-integer column the String binding is unchanged: a spelling the column's +// reader refuses is still the server's own code 53, which withholds as an +// error. +func TestVisible_HostileSpellingsMatchNothing(t *testing.T) { + _, row := parsedRow(t) + + for _, v := range []string{"256", "300", "-1", "1.5", "007", "+7", "7.0", "1e3", "abc", "", " 7", "18446744073709551623"} { + for _, op := range []string{"=", "!=", "<", ">", "in"} { + got, reason := row.VisibleWithReason([]Predicate{{Column: "id", Op: op, Values: []string{v}}}) + assert.False(t, got, "id %s %q", op, v) + assert.Equal(t, ReasonFilter, reason, "id %s %q: answered, not thrown", op, v) + } + } + + for _, v := range []string{"abc", "1.5.5"} { + got, reason := row.VisibleWithReason([]Predicate{{Column: "ratio", Op: "=", Values: []string{v}}}) + assert.False(t, got, "ratio = %q", v) + assert.Equal(t, ReasonError, reason, "ratio = %q", v) + } +} + +// intsTable holds one column per integer width the strict cast has to cover. +func intsTable() *discovery.TableSchema { + return &discovery.TableSchema{ + Name: "ints", + Columns: []discovery.Column{ + {Name: "u8", Type: "UInt8", Position: 1}, + {Name: "u32", Type: "UInt32", Position: 2}, + {Name: "u64", Type: "UInt64", Position: 3}, + {Name: "i64", Type: "Int64", Position: 4}, + {Name: "u128", Type: "UInt128", Position: 5}, + {Name: "i128", Type: "Int128", Position: 6}, + {Name: "u256", Type: "UInt256", Position: 7}, + {Name: "i256", Type: "Int256", Position: 8}, + {Name: "nu64", Type: "Nullable(UInt64)", IsNullable: true, Position: 9}, + }, + } +} + +// intDomain is a column's [min, max]. +func intDomain(typ string) (*big.Int, *big.Int) { + bits := map[string]uint{"8": 8, "32": 32, "64": 64, "128": 128, "256": 256} + pow := func(n uint) *big.Int { return new(big.Int).Lsh(big.NewInt(1), n) } + signed := strings.HasPrefix(typ, "Int") + n := bits[strings.TrimPrefix(strings.TrimPrefix(typ, "U"), "Int")] + if signed { + return new(big.Int).Neg(pow(n - 1)), new(big.Int).Sub(pow(n-1), big.NewInt(1)) + } + return big.NewInt(0), new(big.Int).Sub(pow(n), big.NewInt(1)) +} + +// TestVisible_IntegerClaimsMatchExactlyWhatFits drives every integer width +// with the boundary claims that used to wrap (2^63, 2^64, 2^64+5, 2^127, +// 2^128, 2^255, 2^256, 2^256+5, their negatives) and the non-canonical +// spellings, on every operator, against rows holding 0, 5, the column's MIN +// and MAX (and NULL). The answer must be the mathematical one when the claim +// is the canonical spelling of a value the column can hold, and false +// otherwise — never an over-admit, and never a thrown row. +func TestVisible_IntegerClaimsMatchExactlyWhatFits(t *testing.T) { + eng := TestEngine(t, intsTable()) + tbl, err := eng.Table(tenant.Default, "ints") + require.NoError(t, err) + t.Cleanup(tbl.Release) + + pow := func(n uint) *big.Int { return new(big.Int).Lsh(big.NewInt(1), n) } + add := func(a *big.Int, d int64) *big.Int { return new(big.Int).Add(a, big.NewInt(d)) } + neg := func(a *big.Int) *big.Int { return new(big.Int).Neg(a) } + claims := []string{ + "0", "5", "-1", "255", "256", "4294967295", "4294967296", + "007", "+5", "1.5", "5.0", "1e3", "abc", "", "-0", + } + for _, n := range []*big.Int{ + pow(63), add(neg(pow(63)), -1), neg(pow(63)), add(pow(63), -1), + pow(64), add(pow(64), -1), add(pow(64), 5), pow(127), add(neg(pow(127)), -1), add(pow(127), -1), + pow(128), add(pow(128), -1), pow(255), add(pow(255), -1), neg(pow(255)), pow(256), add(pow(256), -1), add(pow(256), 5), + } { + claims = append(claims, n.String()) + } + + cols := intsTable().Columns + type stored struct { + vals map[string]*big.Int // nil value: NULL + row *Row + } + var rows []stored + for _, label := range []string{"0", "5", "MIN", "MAX"} { + vals := map[string]*big.Int{} + line := make([]any, len(cols)) + for i, c := range cols { + base := strings.TrimSuffix(strings.TrimPrefix(c.Type, "Nullable("), ")") + lo, hi := intDomain(base) + var v *big.Int + switch label { + case "0": + v = big.NewInt(0) + case "5": + v = big.NewInt(5) + case "MIN": + v = lo + case "MAX": + v = hi + } + if c.IsNullable && label == "MIN" { + line[i], vals[c.Name] = nil, nil + continue + } + line[i], vals[c.Name] = v.String(), v + } + b, err := json.Marshal(line) + require.NoError(t, err) + row, err := tbl.ParseRow(tbl.WireColumns, b) + require.NoError(t, err) + t.Cleanup(row.Close) + rows = append(rows, stored{vals: vals, row: row}) + } + + cells := 0 + for _, c := range cols { + base := strings.TrimSuffix(strings.TrimPrefix(c.Type, "Nullable("), ")") + lo, hi := intDomain(base) + for _, claim := range claims { + v, isInt := new(big.Int).SetString(claim, 10) + fits := isInt && v.String() == claim && v.Cmp(lo) >= 0 && v.Cmp(hi) <= 0 + for _, op := range []string{"=", "!=", "<", ">", "in"} { + for _, r := range rows { + cells++ + stored := r.vals[c.Name] + want := false + if fits && stored != nil { + cmp := stored.Cmp(v) + want = map[string]bool{"=": cmp == 0, "in": cmp == 0, "!=": cmp != 0, "<": cmp < 0, ">": cmp > 0}[op] + } + got, reason := r.row.VisibleWithReason([]Predicate{{Column: c.Name, Op: op, Values: []string{claim}}}) + if got != want || (!got && reason != ReasonFilter) { + t.Errorf("%s(%v) %s %q: got %v (%s), want %v", c.Type, stored, op, claim, got, reason, want) + } + } + } + } + } + + // A multi-element _in list keeps the elements that fit and drops the rest, + // element by element: 2^64+5 must not wrap onto the row holding 5. + for _, c := range cols { + for _, r := range rows { + stored := r.vals[c.Name] + want := stored != nil && stored.Sign() == 0 + got := r.row.Visible([]Predicate{{ + Column: c.Name, Op: "in", + Values: []string{add(pow(64), 5).String(), "007", "0", add(pow(256), 5).String(), "abc"}, + }}) + assert.Equal(t, want, got, "%s(%v) IN (2^64+5, 007, 0, 2^256+5, abc)", c.Type, stored) + } + } + t.Logf("%d cells", cells) +} + +// storedRow parses one row whose `tenant` column holds the given value, with +// every other column at sampleRow's value. +func storedRow(t *testing.T, tbl *Table, tenant string) *Row { + t.Helper() + line, err := json.Marshal([]any{7, tenant, "2026-01-15 10:30:00", "12.50", -5, 0.1, []string{"a"}}) + require.NoError(t, err) + row, err := tbl.ParseRow(tbl.WireColumns, line) + require.NoError(t, err) + t.Cleanup(row.Close) + return row +} + +// TestVisible_EscapedStringParamsMatchTheStoredValue pins the shared +// {p:String} encoding (chsql.EscapeStringParam) on the artifact. +// +// Measured on the 26.6 artifact BEFORE the encoding was applied: a stored +// `a\b` compared FALSE against the raw claim value (the artifact's parameter +// reader is ClickHouse's escaped-text reader, so `\b` arrived as a backspace), +// and a stored tab, newline or trailing backslash made the filter fail to +// compile at all — a decline, every row withheld. The SQL path answered true +// for all of them, so the two read surfaces disagreed. With the encoding they +// agree. +func TestVisible_EscapedStringParamsMatchTheStoredValue(t *testing.T) { + eng := TestEngine(t, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + t.Cleanup(tbl.Release) + + cases := []struct { + name string + stored string + }{ + {"embedded backslash", `a\b`}, + {"tab", "a\tb"}, + {"newline", "a\nb"}, + {"carriage return", "a\rb"}, + {"single quote", "O'Brien"}, + {"trailing backslash", `trail\`}, + {"backslash then quote", `a\'b`}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + row := storedRow(t, tbl, tc.stored) + + got, reason := row.VisibleWithReason([]Predicate{{Column: "tenant", Op: "=", Values: []string{tc.stored}}}) + assert.True(t, got, "tenant = %q withheld (reason %q)", tc.stored, reason) + + got, reason = row.VisibleWithReason([]Predicate{{Column: "tenant", Op: "in", Values: []string{"other", tc.stored}}}) + assert.True(t, got, "tenant in (…, %q) withheld (reason %q)", tc.stored, reason) + + got, _ = row.VisibleWithReason([]Predicate{{Column: "tenant", Op: "=", Values: []string{tc.stored + "x"}}}) + assert.False(t, got, "tenant = %q must not match", tc.stored+"x") + }) + } + + // The encoding has to DISTINGUISH, not merely admit: the three characters + // `a\tb` and the three bytes "ab" are different stored values, and each + // must match only its own filter. Unencoded, both filters read as the tab. + t.Run("a literal backslash-t is not a tab", func(t *testing.T) { + literal := storedRow(t, tbl, `a\tb`) + assert.True(t, literal.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{`a\tb`}}})) + assert.False(t, literal.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"a\tb"}}})) + + tab := storedRow(t, tbl, "a\tb") + assert.True(t, tab.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"a\tb"}}})) + assert.False(t, tab.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{`a\tb`}}})) + }) +} + +// TestVisible_PredicatesAreANDed matches the SQL path's AND-joined WHERE. +func TestVisible_PredicatesAreANDed(t *testing.T) { + _, row := parsedRow(t) + + assert.True(t, row.Visible([]Predicate{ + {Column: "tenant", Op: "=", Values: []string{"acme"}}, + {Column: "id", Op: "=", Values: []string{"7"}}, + })) + assert.False(t, row.Visible([]Predicate{ + {Column: "tenant", Op: "=", Values: []string{"acme"}}, + {Column: "id", Op: "=", Values: []string{"8"}}, + })) +} + +func TestVisible_NoPredicatesIsVisible(t *testing.T) { + _, row := parsedRow(t) + assert.True(t, row.Visible(nil)) +} + +// TestVisible_FailsClosed: every shape typelayer cannot answer for must hide +// the row. A row filter that widens on a bad input is the leak class this +// package exists to remove. +func TestVisible_FailsClosed(t *testing.T) { + _, row := parsedRow(t) + + t.Run("unknown column", func(t *testing.T) { + before := row.slot.filters.len() + assert.False(t, row.Visible([]Predicate{{Column: "nosuch", Op: "=", Values: []string{"x"}}})) + assert.Equal(t, before, row.slot.filters.len(), "an unknown column must not reach the compiler") + }) + + t.Run("empty values", func(t *testing.T) { + before := row.slot.filters.len() + assert.False(t, row.Visible([]Predicate{{Column: "tenant", Op: "in", Values: nil}})) + assert.Equal(t, before, row.slot.filters.len(), "an unresolvable claim must not reach the compiler") + }) + + t.Run("unsupported operator", func(t *testing.T) { + assert.False(t, row.Visible([]Predicate{{Column: "tenant", Op: "LIKE", Values: []string{"ac%"}}})) + }) + + t.Run("value the column cannot read", func(t *testing.T) { + // "abc" is not a UInt8 (the strict cast makes it NULL) nor a Float32 + // (the server's own per-row error). Neither is a compile failure. + assert.False(t, row.Visible([]Predicate{{Column: "id", Op: "=", Values: []string{"abc"}}})) + assert.False(t, row.Visible([]Predicate{{Column: "ratio", Op: "=", Values: []string{"abc"}}})) + }) + + t.Run("type mismatch against the column", func(t *testing.T) { + // A String parameter compared with an Array column has no supertype. + assert.False(t, row.Visible([]Predicate{{Column: "tags", Op: "=", Values: []string{"a"}}})) + }) +} + +func TestVisibleWithReason_LabelsTheCause(t *testing.T) { + tbl, row := parsedRow(t) + + ok, reason := row.VisibleWithReason([]Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}}) + assert.True(t, ok) + assert.Empty(t, reason) + + ok, reason = row.VisibleWithReason([]Predicate{{Column: "tenant", Op: "=", Values: []string{"beta"}}}) + assert.False(t, ok) + assert.Equal(t, ReasonFilter, reason) + + ok, reason = row.VisibleWithReason([]Predicate{{Column: "ratio", Op: "=", Values: []string{"abc"}}}) + assert.False(t, ok) + assert.Equal(t, ReasonError, reason, "a value the column's reader refuses throws on the row") + + ok, reason = row.VisibleWithReason([]Predicate{{Column: "id", Op: "=", Values: []string{"abc"}}}) + assert.False(t, ok) + assert.Equal(t, ReasonFilter, reason, "an integer claim that does not fit is answered false") + + ok, reason = row.VisibleWithReason([]Predicate{{Column: "nosuch", Op: "=", Values: []string{"x"}}}) + assert.False(t, ok) + assert.Equal(t, ReasonFilter, reason, "an unexpressible predicate matches nothing, like the SQL path's 1 = 0") + + // A filter that will not compile is the decline case; render never emits + // one, so it is reached through the compiler directly. + assert.Nil(t, tbl.filterOn(row.slot, "notAFunction(`tenant`) = {p0:String}", map[string]string{"p0": "x"})) +} + +// TestParseRow_ColumnsDriftIsAnError: a positional row is only interpretable +// against the generation that produced it. +func TestParseRow_ColumnsDriftIsAnError(t *testing.T) { + eng := TestEngine(t, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + defer tbl.Release() + + _, err = tbl.ParseRow([]string{"id", "tenant"}, []byte(sampleRow)) + require.ErrorIs(t, err, ErrColumnsDrift) + + reordered := append([]string(nil), tbl.WireColumns...) + reordered[0], reordered[1] = reordered[1], reordered[0] + _, err = tbl.ParseRow(reordered, []byte(sampleRow)) + require.ErrorIs(t, err, ErrColumnsDrift, "same names in a different order is still drift") +} + +func TestParseRow_AcceptsALineWithOrWithoutNewline(t *testing.T) { + eng := TestEngine(t, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + defer tbl.Release() + + for _, body := range []string{sampleRow, sampleRow + "\n"} { + row, err := tbl.ParseRow(tbl.WireColumns, []byte(body)) + require.NoError(t, err) + assert.True(t, row.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}})) + row.Close() + } +} + +// TestFilterCache_CompilesOncePerPredicateSet: the cache is what keeps a +// per-subscriber fan-out from recompiling on every event. One Row stays on one +// handle, so the count is that slot's. +func TestFilterCache_CompilesOncePerPredicateSet(t *testing.T) { + _, row := parsedRow(t) + require.Equal(t, 0, row.slot.filters.len()) + + pred := []Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}} + for range 5 { + assert.True(t, row.Visible(pred)) + } + assert.Equal(t, 1, row.slot.filters.len()) + + // A different bound value is a different compiled handle: values are baked + // in at compile time. + assert.False(t, row.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"beta"}}})) + assert.Equal(t, 2, row.slot.filters.len()) +} + +// TestFilterCache_NegativeEntryStopsRecompiling: a broken expression must cost +// one compile and one log line, not one per event. +func TestFilterCache_NegativeEntryStopsRecompiling(t *testing.T) { + tbl, row := parsedRow(t) + for range 3 { + assert.Nil(t, tbl.filterOn(row.slot, "notAFunction(`tenant`) = {p0:String}", map[string]string{"p0": "x"})) + } + assert.Equal(t, 1, row.slot.filters.len()) +} + +// TestFilterCache_BoundedUnderTenantValueChurn: the bound values come from +// tenant claims, so an unbounded cache is a denial of service. Eviction closes +// the handle it drops, against the real library. +func TestFilterCache_BoundedUnderTenantValueChurn(t *testing.T) { + _, row := parsedRow(t) + row.slot.filters = newFilterCache(8) + + for i := range 40 { + pred := []Predicate{{Column: "tenant", Op: "=", Values: []string{strconv.Itoa(i)}}} + assert.False(t, row.Visible(pred)) + } + assert.Equal(t, 8, row.slot.filters.len()) + + // The surviving handles still answer after their neighbours were closed. + assert.True(t, row.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}})) +} + +// TestFilterCache_BudgetIsSplitAcrossTheHandlePool: the 4096 bound is what +// makes the cache not a DoS, and it must be a bound on the TABLE — a pool of +// handles must not multiply it. +func TestFilterCache_BudgetIsSplitAcrossTheHandlePool(t *testing.T) { + eng := TestEngine(t, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + defer tbl.Release() + + // The pool starts at one handle and may grow to its limit; the budget + // holds at the limit, not just at today's size. + p := tbl.pool + assert.Len(t, p.list(), 1, "one handle after Bind; the rest are compiled on contention") + assert.Equal(t, int64(poolSize()), p.limit.Load()) + assert.LessOrEqual(t, p.perSlot*poolSize(), filterCacheSize) + for _, s := range p.list() { + assert.Equal(t, p.perSlot, s.filters.cap) + } +} + +// TestFilterCache_GenerationInvalidatesEntries: a filter only answers for the +// schema handle it was compiled against, so a rebind must not reuse one. +func TestFilterCache_GenerationInvalidatesEntries(t *testing.T) { + eng := TestEngine(t, rowsTable()) + + visible := func() bool { + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + defer tbl.Release() + row, err := tbl.ParseRow(tbl.WireColumns, []byte(sampleRow)) + require.NoError(t, err) + defer row.Close() + return row.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}}) + } + assert.True(t, visible()) + + // Widen a column: the signature changes, so the handle and every filter + // over it are rebuilt, while the wire arity stays the same. + changed := rowsTable() + changed.Columns[0].Type = "UInt16" + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + assert.True(t, visible(), "a rebind recompiles rather than reusing a freed handle") +} + +func (c *filterCache) len() int { + c.mu.Lock() + defer c.mu.Unlock() + return c.order.Len() +} + +// TestVisible_ConcurrentSubscribers: the hub evaluates K filters against one +// parsed block; one LoadedSchema serializes them internally, so this must be +// safe rather than merely fast. +func TestVisible_ConcurrentSubscribers(t *testing.T) { + _, row := parsedRow(t) + + done := make(chan bool, 16) + for i := range 16 { + go func() { + done <- row.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{fmt.Sprintf("t%d", i%4)}}}) + }() + } + for range 16 { + assert.False(t, <-done) + } +} + +// TestTable_ConcurrentAcrossThePool exercises every entry point on one table +// from many goroutines at once. Under -race it is the pin that the pool's +// round-robin, the per-slot filter caches and the shared block parsing are +// safe; a slot chosen per call rather than per Row would show up here as a +// filter and a block on different handles. +func TestTable_ConcurrentAcrossThePool(t *testing.T) { + eng := TestEngine(t, rowsTable()) + + const goroutines = 16 + var wg sync.WaitGroup + for g := range goroutines { + wg.Add(1) + go func() { + defer wg.Done() + for i := range 20 { + tbl, err := eng.Table(tenant.Default, "rows") + if !assert.NoError(t, err) { + return + } + + row, err := tbl.ParseRow(tbl.WireColumns, []byte(sampleRow)) + if assert.NoError(t, err) { + assert.True(t, row.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}})) + assert.False(t, row.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{fmt.Sprintf("t%d", (g+i)%4)}}})) + row.Close() + } + + record := []byte(`{"id":7,"tenant":"acme","ts":"2026-01-15 10:30:00","amt":"12.50","big":-5,"ratio":0.1,"tags":["a"]}` + "\n") + batch, err := tbl.Ingest(FormatJSONEachRow, record) + assert.NoError(t, err) + assert.Len(t, batch.Rows, 1) + + checked, err := tbl.Ingest(FormatJSONEachRow, record, + Predicate{Column: "tenant", Op: "=", Values: []string{"acme"}}) + if assert.NoError(t, err) && assert.Len(t, checked.Rows, 1) { + assert.Empty(t, checked.Rows[0].CheckReason) + assert.NotNil(t, checked.Rows[0].Line) + } + + tbl.Release() + } + }() + } + wg.Wait() +} diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go new file mode 100644 index 00000000..1ecfc854 --- /dev/null +++ b/internal/typelayer/ingest.go @@ -0,0 +1,327 @@ +package typelayer + +import ( + "bytes" + "fmt" + + "github.com/wave-rf/chtypes/go/chtypes" +) + +// InsertSettings are the parsing settings the ingest worker pins on the real +// INSERT, and which chtypes must therefore see — otherwise it answers a +// different question than the server will be asked. Returned fresh so a caller +// may add its own non-parsing pins (async_insert=0) without mutating ours. +func InsertSettings() map[string]string { + return map[string]string{ + "date_time_input_format": "best_effort", + "input_format_null_as_default": "1", + } +} + +// IngestOptions adjusts how IngestWith reads a body. The zero value is +// ClickHouse's own behaviour. +type IngestOptions struct { + // StrictPositional switches header auto-detection off for FormatCSV and + // FormatTSV (input_format_csv_detect_header / input_format_tsv_detect_header + // = 0), so every line is a record. By default ClickHouse consumes a first + // line that names the columns as a header (both settings are on by default; + // measured on the 26.6 server and artifact). Ignored for other formats. + StrictPositional bool +} + +// parseSettings is what a body is parsed under: the insert pins, plus header +// detection off when the caller asked for strict positional CSV/TSV. The worker +// inserts JSONCompactEachRow, so the detect_header settings have no real-INSERT +// twin to keep in step. ok is false for a format Ingest does not parse. +func parseSettings(format Format, opts IngestOptions) (settings map[string]string, ok bool) { + settings = InsertSettings() + switch format { + case FormatJSONEachRow, FormatCSVWithNames, FormatTSVWithNames: + case FormatCSV: + if opts.StrictPositional { + settings["input_format_csv_detect_header"] = "0" + } + case FormatTSV: + if opts.StrictPositional { + settings["input_format_tsv_detect_header"] = "0" + } + default: + return nil, false + } + return settings, true +} + +// RowVerdict is one input record's answer, index-aligned with the records the +// caller wrote into the body. +type RowVerdict struct { + Accepted bool + // Code is ClickHouse's own error code when the record was refused (27, 117, + // 6 …) and 0 otherwise. Declined verdicts carry no code: chtypes did not + // answer, so there is nothing to attribute to the data. + Code int + Message string + // Declined marks "the validation engine could not answer", never "the data + // is bad". A caller must not turn it into a 400. + Declined bool + // CheckReason is why the insert checks given to Ingest did not admit an + // accepted record: ReasonFilter (the data says no), ReasonError or + // ReasonDecline (we could not tell). "" when they admitted it or there were + // none. A record with a CheckReason has no Line — only the rows the filter + // admits are exported — and Message may carry the predicate's own error. + CheckReason string + // Line is the record as ClickHouse's own JSONCompactEachRow writer + // serialized it, without the trailing newline. nil unless Accepted with no + // CheckReason. It is a sub-slice of the batch's exported bytes, so a caller + // that outlives the request must copy it. + Line []byte +} + +// Batch holds one verdict per input record, in input order. +type Batch struct { + // Rows holds one verdict per record chtypes read, in input order. When + // Answered is false chtypes gave no per-record detail (the whole batch was + // declined) and Rows is padded to the body's line count so a caller still + // has something index-shaped to report. + Rows []RowVerdict + Answered bool + // Refused is ClickHouse's own refusal of a WithNames body as a whole — a + // header naming a column the schema does not have, or naming one twice — + // before any record was read. Rows is then empty; nil otherwise. + Refused *Refusal +} + +// Refusal is ClickHouse's verdict on a body rather than on any one record. +type Refusal struct { + Code int + Message string +} + +// Ingest asks ClickHouse's own parser whether each record in body would +// insert, in ONE call, and exports the accepted rows as JSONCompactEachRow. +// +// format is how body is spelled: +// +// - FormatJSONEachRow: newline-separated and name-addressed (an NDJSON body +// is byte-identical to this); +// - FormatCSV, FormatTSV: positional in declaration order. ClickHouse's +// header auto-detection applies (a first line naming the columns is +// consumed, not a record) unless IngestWith sets StrictPositional, where a +// header line is one failed record; +// - FormatCSVWithNames, FormatTSVWithNames: a first line naming the columns +// in any order. It is not a record, so Rows index the data lines; a column +// it omits takes its DEFAULT, and one it names that the schema lacks (or +// names twice) refuses the whole body (Batch.Refused, code 117). +// +// Any other format is a programming error and returns an error without +// touching the data — a binding must never declare a format it has not asked +// the artifact about. +// +// checks are a role's insert check clauses. They compile to ONE filter, +// AND-joined, attached to the same parse (RowsExportWith; the chtypes SDK's +// filters guide, "Exporting only the rows a filter admits"), so each record's +// parse outcome and check answer come from one read of the body and only the +// admitted records are exported. The parse outcome decides first: a record +// chtypes did not accept reports its own error whatever the filter says (it +// answers such a row 'd'). A predicate with no Values matches nothing and +// never reaches the compiler; a filter that will not compile declines every +// accepted record. The compiled filter is cached per (generation, expression, +// values) on the handle it runs against. +// +// Document flags stay lean (verdicts and exported bytes only). The per-value +// provenance DocValues would give costs 2.65× on this path and nothing here +// reads it. +// +// The returned error is for a Go-level failure only — every data verdict is in +// the batch. +func (t *Table) Ingest(format Format, body []byte, checks ...Predicate) (Batch, error) { + return t.IngestWith(format, IngestOptions{}, body, checks...) +} + +// IngestWith is Ingest with options; see IngestOptions. +func (t *Table) IngestWith(format Format, opts IngestOptions, body []byte, checks ...Predicate) (Batch, error) { + settings, ok := parseSettings(format, opts) + if !ok { + return Batch{}, fmt.Errorf( + "typelayer: Ingest cannot parse format %d; use FormatJSONEachRow, FormatCSV, FormatTSV, FormatCSVWithNames or FormatTSVWithNames", + int(format)) + } + if t.pool == nil { + return Batch{}, &Unavailable{Tenant: t.tenant, Table: t.Name, Cause: t.cause} + } + // The filter must be compiled on the handle the parse runs on: one from + // another handle rejects the whole call. + s := t.pool.acquire() + defer t.pool.release(s) + filter, uniform := t.checkFilter(s, checks) + res, err := export(s, format, body, settings, filter) + if err != nil && filter != nil { + // The cached filter was evicted and closed between lookup and use. Fail + // the checks closed, as an evaluation error would, not the request. + filter, uniform = nil, ReasonDecline + res, err = export(s, format, body, settings, nil) + } + if err != nil { + return Batch{}, err + } + + // Only a fully accepted batch exports bytes. Gate on the outcome, never on + // RowsPassed, which counts the admitted rows of a rejected batch too. + if res.Outcome != chtypes.Accepted { + if len(res.Rows) == 0 && res.Outcome == chtypes.Rejected && res.ErrCode != 0 && + (format == FormatCSVWithNames || format == FormatTSVWithNames) { + return Batch{Answered: true, Refused: &Refusal{Code: res.ErrCode, Message: res.ErrMsg}}, nil + } + return declineAll(countRecords(body, len(res.Rows)), firstNonEmpty(res.ExportDeclined, res.ErrMsg, res.Outcome.String())), nil + } + // Accepted but withheld (the full-arity guard, a serialization failure): + // nothing can be forwarded, whatever the per-row detail says. + if res.ExportDeclined != "" { + return declineAll(countRecords(body, len(res.Rows)), res.ExportDeclined), nil + } + + // chtypes answered per record, so its count is the record count: the + // verdicts are index-aligned with the records it read, and padding to the + // body's newline count would invent declined records out of blank lines + // and pretty-printed framing. + out := Batch{Rows: make([]RowVerdict, len(res.Rows)), Answered: true} + for i := range out.Rows { + v := rowVerdict(res.Rows[i], span(res, i), filter != nil) + if v.Accepted && uniform != "" { + v.Line, v.CheckReason = nil, uniform + } + out.Rows[i] = v + } + return out, nil +} + +// checkFilter resolves the check clauses to the filter attached to the parse, +// or to the one answer every accepted record gets when no filter runs: +// ReasonFilter for a predicate render cannot express (it matches nothing, like +// the SQL path's `1 = 0`), ReasonDecline for one that will not compile. +func (t *Table) checkFilter(s *schemaSlot, checks []Predicate) (*chtypes.LoadedFilter, string) { + if len(checks) == 0 { + return nil, "" + } + expr, params, ok := t.render(checks) + if !ok { + return nil, ReasonFilter + } + if f := t.filterOn(s, expr, params); f != nil { + return f, "" + } + return nil, ReasonDecline +} + +// export is the one parse. With no filter RowsExportWith is RowsExport. +func export(s *schemaSlot, format Format, body []byte, settings map[string]string, f *chtypes.LoadedFilter) (chtypes.BatchResult, error) { + if f == nil { + return s.schema.RowsExportWith(format, body, settings, chtypes.JSONCompactEachRow) + } + return s.schema.RowsExportWith(format, body, settings, chtypes.JSONCompactEachRow, chtypes.WithRowFilter(f)) +} + +// rowVerdict maps one chtypes RowResult. An unsupported setting is the engine +// declining even when the row itself parsed, so it is checked before the +// outcome, and the outcome before the filter's verdict. +func rowVerdict(r chtypes.RowResult, line []byte, filtered bool) RowVerdict { + if len(r.UnsupportedSettings) > 0 { + return RowVerdict{Declined: true, Message: "chtypes does not support setting(s) " + joinQuoted(r.UnsupportedSettings)} + } + switch r.Outcome { + case chtypes.Accepted: + if filtered { + // A nil verdict is a row the filter never answered: it must not pass. + if r.Verdict == nil { + return RowVerdict{Accepted: true, CheckReason: ReasonDecline} + } + if ok, reason := verdictBool(*r.Verdict); !ok { + return RowVerdict{Accepted: true, CheckReason: reason, Message: r.VerdictErr} + } + } + if line == nil { + return RowVerdict{Declined: true, Message: "accepted but no bytes were exported for this row"} + } + return RowVerdict{Accepted: true, Line: line} + case chtypes.Skipped, chtypes.Rejected: + // ErrCode/ErrMsg, not VerdictCode/VerdictErr: chtypes answers such a row + // 'd', and on 26.6 leaves the verdict's own code and message empty. + return RowVerdict{Code: r.ErrCode, Message: r.ErrMsg} + case chtypes.Unsupported, chtypes.AcceptedPoisoned: + // AcceptedPoisoned holds a value no writer can honestly serialize, so + // like Unsupported it yields no bytes and is not a data verdict. + return declinedVerdict(r) + default: + return declinedVerdict(r) + } +} + +func firstNonEmpty(s ...string) string { + for _, v := range s { + if v != "" { + return v + } + } + return "" +} + +// declinedVerdict reports an outcome that is not a verdict about the data, +// keeping the engine's own wording when it supplied any. +func declinedVerdict(r chtypes.RowResult) RowVerdict { + msg := r.ErrMsg + if msg == "" { + msg = r.Outcome.String() + } + return RowVerdict{Declined: true, Message: msg} +} + +// span slices row i's exported line out of the payload, dropping the trailing +// newline the writer emits. Returns nil when the row contributed no bytes. +func span(res chtypes.BatchResult, i int) []byte { + if i >= len(res.Spans) { + return nil + } + s := res.Spans[i] + if s.Len <= 0 || s.Off < 0 || s.Off+s.Len > len(res.Payload) { + return nil + } + return bytes.TrimSuffix(res.Payload[s.Off:s.Off+s.Len], []byte("\n")) +} + +func declineAll(n int, msg string) Batch { + b := Batch{Rows: make([]RowVerdict, n)} + for i := range b.Rows { + b.Rows[i] = RowVerdict{Declined: true, Message: msg} + } + return b +} + +// countRecords recovers the input record count when chtypes returned no +// per-row detail, so the caller still gets an index-aligned answer. JSONEachRow +// records are newline-separated and json.Marshal escapes any newline inside a +// value, so counting lines is exact for the bodies this package is handed. A +// CSV field may legally contain a raw newline, and a WithNames header is a +// line but not a record, so for those formats the fallback can OVER-count, +// which produces extra declined verdicts — never an extra acceptance. +func countRecords(body []byte, known int) int { + if known > 0 { + return known + } + n := bytes.Count(body, []byte("\n")) + if len(body) > 0 && !bytes.HasSuffix(body, []byte("\n")) { + n++ + } + return n +} + +func joinQuoted(names []string) string { + var b bytes.Buffer + for i, n := range names { + if i > 0 { + b.WriteString(", ") + } + b.WriteByte('"') + b.WriteString(n) + b.WriteByte('"') + } + return b.String() +} diff --git a/internal/typelayer/ingest_test.go b/internal/typelayer/ingest_test.go new file mode 100644 index 00000000..ddc133c3 --- /dev/null +++ b/internal/typelayer/ingest_test.go @@ -0,0 +1,212 @@ +package typelayer + +import ( + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +func ingestTable(t *testing.T) *Table { + t.Helper() + eng := TestEngine(t, eventsTable()) + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + t.Cleanup(tbl.Release) + return tbl +} + +// TestIngest_PerRecordVerdicts pins the measured behaviour of the whole +// compile profile at once: a bad value is one record's rejection rather than +// the batch's, the codes are ClickHouse's own, and the accepted records come +// back as the bytes the server itself would store. +func TestIngest_PerRecordVerdicts(t *testing.T) { + tbl := ingestTable(t) + + body := strings.Join([]string{ + `{"id":1,"name":"a","ts":"2026-01-15 10:30:00.000","tags":["x"],"score":null}`, + `{"id":"bad","name":"b","ts":"2026-01-15 10:30:00.000","tags":[],"score":1}`, + `{"id":3,"name":"c","ts":"2026-01-15 10:30:00.000","tags":[],"score":null,"extra":1}`, + `{"id":4,"name":"d","ts":"2026-01-15 10:30:00.000","tags":[],"score":null}`, + }, "\n") + "\n" + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(body)) + require.NoError(t, err) + require.Len(t, batch.Rows, 4, "one verdict per input record, index-aligned") + + assert.True(t, batch.Rows[0].Accepted) + assert.Equal(t, 0, batch.Rows[0].Code) + + // An unparseable value: ClickHouse's own CANNOT_PARSE_TEXT. + assert.False(t, batch.Rows[1].Accepted) + assert.False(t, batch.Rows[1].Declined) + assert.Equal(t, 27, batch.Rows[1].Code) + assert.NotEmpty(t, batch.Rows[1].Message) + + // input_format_skip_unknown_fields=0 turns an unknown field into a real + // per-row rejection instead of silent data loss. + assert.False(t, batch.Rows[2].Accepted) + assert.Equal(t, 117, batch.Rows[2].Code) + assert.Contains(t, batch.Rows[2].Message, "extra") + + assert.True(t, batch.Rows[3].Accepted) +} + +// TestIngest_AcceptedLineIsOneWireRow: the published line must be exactly one +// JSONCompactEachRow row with no trailing newline, with the MATERIALIZED column +// absent and the volatile DEFAULT already baked in. +func TestIngest_AcceptedLineIsOneWireRow(t *testing.T) { + tbl := ingestTable(t) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":7,"name":"a","ts":"2026-01-15 10:30:00.000","tags":["x"],"score":null}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + + line := string(batch.Rows[0].Line) + assert.NotContains(t, line, "\n") + assert.True(t, strings.HasPrefix(line, "["), line) + assert.True(t, strings.HasSuffix(line, "]"), line) + assert.Equal(t, len(tbl.WireColumns), strings.Count(line, ",")+1, "one cell per wire column: %s", line) + // now() was substituted at parse time, so the worker inserts the bytes + // verbatim and the server never re-evaluates the clock. + assert.NotContains(t, line, "now()") +} + +// TestIngest_OverflowIsStoredTruth, not the producer's spelling: 256 into a +// UInt8 wraps to 0, which is what the table will hold and therefore what the +// stream must show (#372's payload-vs-stored asymmetry, closed by construction). +func TestIngest_OverflowIsStoredTruth(t *testing.T) { + tbl := ingestTable(t) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":256,"name":"a","ts":"2026-01-15 10:30:00.000","tags":[],"score":null}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.True(t, strings.HasPrefix(string(batch.Rows[0].Line), "[0,"), string(batch.Rows[0].Line)) +} + +// TestIngest_ComputedColumnsAreRejectedPerRecord: a record naming a +// MATERIALIZED, ALIAS or EPHEMERAL column gets ClickHouse's own 117 and the +// rest of the batch still gets verdicts — no WaveHouse-side guard needed. +func TestIngest_ComputedColumnsAreRejectedPerRecord(t *testing.T) { + schema := &discovery.TableSchema{ + Name: "computed", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt32", Position: 1}, + {Name: "e", Type: "UInt8", DefaultKind: "EPHEMERAL", HasDefault: true, Position: 2}, + {Name: "d", Type: "UInt8", DefaultKind: "DEFAULT", DefaultExpression: "e + 1", HasDefault: true, Position: 3}, + {Name: "a", Type: "UInt8", DefaultKind: "ALIAS", DefaultExpression: "id + 2", HasDefault: true, Position: 4}, + }, + } + eng := TestEngine(t, schema) + tbl, err := eng.Table(tenant.Default, "computed") + require.NoError(t, err) + defer tbl.Release() + + assert.Equal(t, []string{"id", "d"}, tbl.WireColumns) + + body := strings.Join([]string{ + `{"id":1}`, + `{"id":2,"e":5}`, + `{"id":3,"a":9}`, + `{"id":4,"d":7}`, + }, "\n") + "\n" + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(body)) + require.NoError(t, err) + require.Len(t, batch.Rows, 4) + + assert.True(t, batch.Rows[0].Accepted) + assert.Equal(t, 117, batch.Rows[1].Code) + assert.Contains(t, batch.Rows[1].Message, "e") + assert.Equal(t, 117, batch.Rows[2].Code) + assert.Contains(t, batch.Rows[2].Message, "a") + assert.True(t, batch.Rows[3].Accepted) + // ClickHouse's writer separates cells with ", " — the worker inserts these + // bytes verbatim, so nothing may re-render them. + assert.Equal(t, "[4, 7]", string(batch.Rows[3].Line)) +} + +// TestInsertSettings_AreAllSupported: every setting typelayer passes must be one +// chtypes understands. An unsupported one turns a real verdict into a decline, +// which is a silent availability regression on the ingest path. +func TestInsertSettings_AreAllSupported(t *testing.T) { + tbl := ingestTable(t) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":1,"name":"a","ts":"2026-01-15 10:30:00.000","tags":[],"score":null}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + require.False(t, batch.Rows[0].Declined, "InsertSettings must not contain a setting chtypes rejects: %s", batch.Rows[0].Message) + assert.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) +} + +// TestInsertSettings_NullAsDefaultApplies: the worker pins it on the real +// INSERT, so chtypes has to see the same rule — an explicit null in a +// non-nullable column becomes the type's default rather than a rejection. +func TestInsertSettings_NullAsDefaultApplies(t *testing.T) { + tbl := ingestTable(t) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":null,"name":"a","ts":"2026-01-15 10:30:00.000","tags":[],"score":null}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + assert.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.True(t, strings.HasPrefix(string(batch.Rows[0].Line), "[0,"), string(batch.Rows[0].Line)) +} + +// TestInsertSettings_BestEffortDateTime: without the worker's own +// date_time_input_format pin, chtypes would reject an RFC3339 value the real +// server accepts — a manufactured over-reject. +func TestInsertSettings_BestEffortDateTime(t *testing.T) { + tbl := ingestTable(t) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":1,"name":"a","ts":"2026-01-15T10:30:00Z","tags":[],"score":null}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + assert.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) +} + +func TestInsertSettings_ReturnsAFreshMap(t *testing.T) { + t.Parallel() + a := InsertSettings() + a["async_insert"] = "0" + assert.NotContains(t, InsertSettings(), "async_insert") + assert.Equal(t, map[string]string{ + "date_time_input_format": "best_effort", + "input_format_null_as_default": "1", + }, InsertSettings()) +} + +// TestIngest_CountIsChtypesOwn: blank lines and pretty-printed framing are +// not records. chtypes' per-record answer is the record count; nothing is +// padded on top of it. +func TestIngest_CountIsChtypesOwn(t *testing.T) { + tbl := ingestTable(t) + + ndjson := "{\"id\":1,\"name\":\"a\",\"ts\":\"2026-01-15 10:30:00.000\",\"tags\":[],\"score\":null}\n\n" + + "{\"id\":2,\"name\":\"b\",\"ts\":\"2026-01-15 10:30:00.000\",\"tags\":[],\"score\":null}\n" + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(ndjson)) + require.NoError(t, err) + assert.True(t, batch.Answered) + require.Len(t, batch.Rows, 2, "a blank line is not a record") + assert.True(t, batch.Rows[0].Accepted) + assert.True(t, batch.Rows[1].Accepted) + + pretty := "{\n \"id\": 3,\n \"name\": \"c\",\n \"ts\": \"2026-01-15 10:30:00.000\",\n \"tags\": [],\n \"score\": null\n}\n" + batch, err = tbl.Ingest(FormatJSONEachRow, []byte(pretty)) + require.NoError(t, err) + require.Len(t, batch.Rows, 1, "a pretty-printed object is one record, not one per line") + assert.True(t, batch.Rows[0].Accepted) +} + +func TestCountRecords(t *testing.T) { + t.Parallel() + assert.Equal(t, 0, countRecords(nil, 0)) + assert.Equal(t, 1, countRecords([]byte(`{"a":1}`), 0)) + assert.Equal(t, 2, countRecords([]byte("{}\n{}\n"), 0)) + assert.Equal(t, 2, countRecords([]byte("{}\n{}"), 0)) + assert.Equal(t, 5, countRecords([]byte("{}\n{}"), 5), "chtypes' own count wins when it has one") +} diff --git a/internal/typelayer/pool.go b/internal/typelayer/pool.go new file mode 100644 index 00000000..28f7cd98 --- /dev/null +++ b/internal/typelayer/pool.go @@ -0,0 +1,211 @@ +package typelayer + +import ( + "errors" + "fmt" + "log/slog" + "runtime" + "slices" + "sync" + "sync/atomic" + + "github.com/wave-rf/chtypes/go/chtypes" +) + +// maxPoolSize caps the identically-compiled handles a base table grows to. +// +// A LoadedSchema serializes its own calls, so one handle per table is a +// ceiling. Measured on a Linux arm64 VM (BenchmarkIngest_HandlePool, artifacts +// 26.6 and 25.8), own handles scale 4-6x at 8 goroutines, while a +// process-wide serialization gate never beats one thread (0.83-1.0x). A +// compiled handle costs ~40 KiB (~96 KiB warm). That cost multiplies by +// tables and by tenants, so a pool starts at one handle and grows only when +// every handle it has is busy (see pool.acquire). Role tables keep one handle +// (roleHandles): hot tenants already spread over distinct role handles, and +// 256 shapes x 8 warm handles would be ~200 MiB. An earlier darwin run showed +// a flat curve, so treat the size as hardware-dependent and re-measure it on +// the deployment hardware. +const maxPoolSize = 8 + +// roleHandles is the handle count of a role-shape table. +const roleHandles = 1 + +// poolSize is how many identical handles one base-table shape may grow to. +func poolSize() int { + return max(min(runtime.GOMAXPROCS(0), maxPoolSize), 1) +} + +// compileSettings is the fixed parsing profile every table handle is compiled +// with. allow_errors_ratio turns "first bad row ends the batch" into +// skip-and-continue so every record gets its own verdict; skip_unknown_fields=0 +// makes an unknown field a real per-row rejection (ClickHouse code 117) instead +// of silent data loss. Neither is ever forwarded to the real INSERT. +var compileSettings = map[string]string{ + "input_format_allow_errors_ratio": "1", + "input_format_skip_unknown_fields": "0", +} + +// schemaSlot is one compiled handle plus the filters compiled against it. A +// filter is bound to the schema it was compiled from — evaluating it against a +// block from another handle takes BOTH handles' locks — so each slot owns its +// own cache and one evaluation stays on one slot. +type schemaSlot struct { + schema *chtypes.LoadedSchema + filters *filterCache + // busy counts the calls (and parsed Rows) currently on this slot. + busy atomic.Int32 +} + +// closeSlots tears handles down. Filters go before the schema, which the C +// layer requires, and the cache index is emptied so no freed pointer survives. +func closeSlots(slots []*schemaSlot) { + for _, s := range slots { + if s == nil { + continue + } + if s.filters != nil { + s.filters.closeAll() + } + if s.schema != nil { + s.schema.Close() + } + } +} + +// pool is one table shape's identically-compiled handles. It is compiled with +// one handle and grows lazily, up to its limit, when a call finds every handle +// busy. Every slot is compiled from the same declaration list, so the handles +// are interchangeable and a call may land on any of them. +// +// Lifetime: a pool is only used by callers holding its Table's read lock, and +// only closed under that Table's write lock or once nothing can reach it, so +// growth and close never overlap. +type pool struct { + lib *chtypes.Library + ddl string + perSlot int // filter-cache capacity of each slot + // limit is the most slots the pool will hold; lowered to the current size + // if a growth compile is ever refused, so a refusal costs one compile. + limit atomic.Int64 + // grow admits one growing caller at a time; the others share a busy slot + // rather than wait for a compile. + grow sync.Mutex + // slots is replaced, never mutated, so readers need no lock. + slots atomic.Pointer[[]*schemaSlot] + next atomic.Uint64 +} + +// newPool compiles a pool's first handle. A refusal is the whole shape's: the +// handles are interchangeable by construction, so there is no half-pool. +// +// The per-table filter budget is SPLIT across the pool's limit rather than +// multiplied by it: values are tenant-controlled and baked into a compiled +// handle, so the bound that makes the cache not-a-DoS has to be a bound on +// the table. +func newPool(lib *chtypes.Library, ddl string, limit int) (*pool, string) { + limit = max(limit, 1) + p := &pool{lib: lib, ddl: ddl, perSlot: max(filterCacheSize/limit, 1)} + p.limit.Store(int64(limit)) + first, err := p.compile() + if err != nil { + return nil, describeCompileError(err) + } + p.slots.Store(&[]*schemaSlot{first}) + return p, "" +} + +func (p *pool) compile() (*schemaSlot, error) { + schema, err := p.lib.CompileDDL(p.ddl, chtypes.WithCompileSettings(compileSettings)) + if err != nil { + return nil, err + } + return &schemaSlot{schema: schema, filters: newFilterCache(p.perSlot)}, nil +} + +// list is the current slots, oldest first. +func (p *pool) list() []*schemaSlot { return *p.slots.Load() } + +// first is the slot every pool has: what column introspection reads. +func (p *pool) first() *schemaSlot { return p.list()[0] } + +// acquire picks the slot one call (or one parsed Row) runs on and marks it +// busy until release. An idle slot is taken first; when every slot is busy and +// the pool is below its limit, the caller compiles one more and takes it; a +// caller that cannot grow the pool shares a busy slot, whose handle serializes +// the two calls. Growing on contention rather than at Bind is what keeps a +// quiet table, of a quiet tenant, at one handle. +func (p *pool) acquire() *schemaSlot { + slots := p.list() + n := uint64(len(slots)) + start := p.next.Add(1) % n + for i := range n { + if s := slots[(start+i)%n]; s.busy.CompareAndSwap(0, 1) { + return s + } + } + if int64(len(slots)) < p.limit.Load() && p.grow.TryLock() { + s := p.growLocked() + p.grow.Unlock() + if s != nil { + return s + } + } + s := slots[start] + s.busy.Add(1) + return s +} + +// growLocked returns a slot already marked busy for the caller — one that +// became idle, or one more compiled — or nil when the pool is full or the +// compile is refused. +func (p *pool) growLocked() *schemaSlot { + cur := p.list() + // A slot freed (or grown) since the caller looked is cheaper than a compile. + for _, s := range cur { + if s.busy.CompareAndSwap(0, 1) { + return s + } + } + if int64(len(cur)) >= p.limit.Load() { + return nil + } + s, err := p.compile() + if err != nil { + // The first handle compiled from this same list, so this is not a + // verdict about the table: stop growing and keep serving on what the + // pool has. + p.limit.Store(int64(len(cur))) + slog.Warn("chtypes refused an extra handle for a table; it keeps the handles it has", + "handles", len(cur), "cause", describeCompileError(err)) + return nil + } + s.busy.Store(1) + next := append(slices.Clip(cur), s) + p.slots.Store(&next) + return s +} + +// release ends a call acquire started. +func (p *pool) release(s *schemaSlot) { s.busy.Add(-1) } + +// close tears every handle down. See pool's lifetime note. +func (p *pool) close() { + if p != nil { + closeSlots(p.list()) + } +} + +// describeCompileError keeps ClickHouse's own refusal (a real code) distinct +// from chtypes declining to answer — the two mean different things to an +// operator and to the HTTP status a caller picks. +func describeCompileError(err error) string { + var se *chtypes.SchemaError + if errors.As(err, &se) { + return fmt.Sprintf("compile refused with ClickHouse code %d: %s", se.Code, se.Msg) + } + var ue *chtypes.UnsupportedError + if errors.As(err, &ue) { + return "chtypes declined: " + ue.Msg + } + return err.Error() +} diff --git a/internal/typelayer/roletable.go b/internal/typelayer/roletable.go new file mode 100644 index 00000000..0664defe --- /dev/null +++ b/internal/typelayer/roletable.go @@ -0,0 +1,315 @@ +package typelayer + +import ( + "container/list" + "crypto/sha256" + "fmt" + "log/slog" + "maps" + "slices" + "strings" + "sync" + + "github.com/wave-rf/chtypes/go/chtypes" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// roleCacheSize bounds the per-role shapes held per table. A shape's Defaults +// values come from tenant claims and are baked into the compiled handle, so an +// unbounded cache is a memory and CPU denial of service — the same reason +// filterCache is bounded. Each entry costs roleHandles (1) compiled handle. +const roleCacheSize = 256 + +// RoleShape is the projection of a table a role may insert through. It is the +// whole input to Engine.RoleTable and the whole cache key, so two roles with +// the same shape share one compiled handle. +// +// Columns is the set of columns the role may write. nil means "every column" +// and is the identity shape; a non-nil, EMPTY slice means "no column", which +// compiles to nothing and fails closed. Order is irrelevant — the DDL always +// follows the table's own declaration order — and a name the table does not +// have is ignored. Columns the role cannot supply anyway (MATERIALIZED, ALIAS, +// EPHEMERAL) are always kept: dropping one would change what the server +// computes, and a MATERIALIZED expression over a dropped column would not +// compile at all. +// +// Defaults maps a column to a literal value injected when a record omits it, +// rendered into the column's DEFAULT clause. A value the record DOES supply +// still wins (measured on the 26.6 artifact; see +// TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue). Every key +// must name a column the shape +// keeps and must be an ordinary column (no DEFAULT, or a plain DEFAULT) — +// anything else is a programming error and returns an error rather than +// silently reshaping the table. +type RoleShape struct { + Columns []string + Defaults map[string]string +} + +// identity reports whether the shape asks for nothing the base table does not +// already answer, in which case no second handle is compiled. +func (s RoleShape) identity() bool { + return s.Columns == nil && len(s.Defaults) == 0 +} + +// key is the cache key: the generation that owns the schema plus a canonical +// hash of the shape. Every component is length-prefixed, so no column name or +// claim value can spell another shape's encoding. +func (s RoleShape) key(generation uint64) string { + var b strings.Builder + fmt.Fprintf(&b, "g%d\x1e", generation) + if s.Columns == nil { + b.WriteString("*\x1e") + } else { + cols := slices.Clone(s.Columns) + slices.Sort(cols) + cols = slices.Compact(cols) + for _, c := range cols { + fmt.Fprintf(&b, "%d:%s", len(c), c) + } + b.WriteByte(0x1e) + } + for _, k := range slices.Sorted(maps.Keys(s.Defaults)) { + v := s.Defaults[k] + fmt.Fprintf(&b, "%d:%s%d:%s", len(k), k, len(v), v) + } + // Hashed rather than kept verbatim: the Defaults values are tenant claims, + // so the key's size must not be the caller's to choose. + sum := sha256.Sum256([]byte(b.String())) + return string(sum[:]) +} + +// RoleTable returns the compiled handle for one role's projection of a table, +// read-locked exactly like Engine.Table: the caller must Release it, and the +// handle stays alive and stable until it does. +// +// The shape is answered by ClickHouse's own parser rather than by a Go walk +// over the record's keys: a column the role may not write is +// simply absent from the compiled DDL, so a record naming it is refused +// per-row with ClickHouse's own code 117 "Unknown field found while parsing +// JSONEachRow format: x"; a Defaults column is declared DEFAULT '', +// quoted by the library's own QuoteLiteral, so an absent value is filled and a +// supplied one still wins. +// +// The identity shape (no column restriction, no defaults) returns the base +// table itself — no second handle, no cache entry. +// +// A shape that does not compile is cached as a negative entry and reported as +// *Unavailable, so a broken policy costs one compile and one log line per +// generation rather than one per request. +func (e *Engine) RoleTable(id tenant.ID, table string, shape RoleShape) (*Table, error) { + base, err := e.Table(id, table) + if err != nil { + return nil, err + } + if shape.identity() { + return base, nil // still read-locked; the caller's Release covers it + } + rt, evicted, err := base.roleTable(shape) + base.Release() + // Closed after the cache lock and the base read lock are both gone: it + // waits for the evicted shape's own readers, which are requests in flight. + if evicted != nil { + evicted.close("evicted from the role cache") + } + if err != nil { + return nil, err + } + return rt, nil +} + +// roleTable is RoleTable's cache half, run under the base table's read lock. +// It returns the projection read-locked, plus the entry its insertion evicted +// (to be closed by the caller, outside the cache lock). +func (t *Table) roleTable(shape RoleShape) (*Table, *Table, error) { + key := shape.key(t.Generation) + c := t.roles + c.mu.Lock() + defer c.mu.Unlock() + + if el, hit := c.index[key]; hit { + c.order.MoveToFront(el) + e := el.Value.(*roleEntry) + if e.table == nil { + return nil, nil, &Unavailable{Tenant: t.tenant, Table: t.Name, Cause: e.cause} + } + // Taken while the base read lock is still held, so a rebind cannot be + // closing this projection underneath us. + e.table.mu.RLock() + return e.table, nil, nil + } + + rt, cause := t.compileRole(shape) + if cause != "" { + slog.Error("chtypes could not compile a per-role schema; every insert for this role fails closed", + "tenant", t.tenant, "table", t.Name, "generation", t.Generation, + "allowed_columns", shape.Columns, "default_columns", slices.Sorted(maps.Keys(shape.Defaults)), + "cause", cause) + } + el := c.order.PushFront(&roleEntry{key: key, table: rt, cause: cause}) + c.index[key] = el + var evicted *Table + if c.order.Len() > c.cap { + evicted = c.evictOldestLocked() + } + if rt == nil { + return nil, evicted, &Unavailable{Tenant: t.tenant, Table: t.Name, Cause: cause} + } + rt.mu.RLock() + return rt, evicted, nil +} + +// compileRole builds and compiles the role's declaration list. It returns +// (nil, cause) for every refusal. +func (t *Table) compileRole(shape RoleShape) (*Table, string) { + cols, wire, err := roleColumns(t.lib, t.discovered, shape) + if err != nil { + return nil, err.Error() + } + ddl, rerr := t.lib.ReconstructDDL(cols) + if rerr != nil { + return nil, "cannot reconstruct role column declarations: " + rerr.Error() + } + p, cause := newPool(t.lib, ddl, roleHandles) + if cause != "" { + return nil, cause + } + schema := p.first().schema + declared, cause := declaredColumns(t.lib, schema, nil) + if cause != "" { + p.close() + return nil, cause + } + + return &Table{ + Name: t.Name, + Generation: t.Generation, + WireColumns: deriveWireColumns(schema, wire), + tenant: t.tenant, + pool: p, + cols: declared, + lib: t.lib, + }, "" +} + +// roleColumns projects the table's discovered columns onto a shape. It returns +// the declaration list and the wire column names (the declaration list minus +// the three kinds a positional INSERT never carries). An injected value is +// quoted by lib.QuoteLiteral, ClickHouse's own quoteString, so it reaches the +// compiler as one string literal whatever bytes it holds. +func roleColumns(lib *chtypes.Library, src []discovery.Column, shape RoleShape) ([]chtypes.DiscoveredColumn, []string, error) { + var allowed map[string]struct{} + if shape.Columns != nil { + allowed = make(map[string]struct{}, len(shape.Columns)) + for _, c := range shape.Columns { + allowed[c] = struct{}{} + } + } + + cols := make([]chtypes.DiscoveredColumn, 0, len(src)) + wire := make([]string, 0, len(src)) + kept := make(map[string]discovery.Column, len(src)) + for _, c := range src { + computed := c.DefaultKind == "MATERIALIZED" || c.DefaultKind == "ALIAS" || c.DefaultKind == "EPHEMERAL" + if allowed != nil && !computed { + if _, ok := allowed[c.Name]; !ok { + continue + } + } + dc := chtypes.DiscoveredColumn{ + Name: c.Name, + Type: c.Type, + DefaultKind: c.DefaultKind, + DefaultExpression: c.DefaultExpression, + Position: c.Position, + } + if v, inject := shape.Defaults[c.Name]; inject { + if computed { + return nil, nil, fmt.Errorf( + "cannot inject a default into column %q: it is %s", c.Name, c.DefaultKind) + } + lit, err := lib.QuoteLiteral(v) + if err != nil { + return nil, nil, fmt.Errorf("cannot quote the default for column %q: %w", c.Name, err) + } + dc.DefaultKind, dc.DefaultExpression = "DEFAULT", lit + } + cols = append(cols, dc) + if !computed { + wire = append(wire, c.Name) + } + kept[c.Name] = c + } + + // A default for a column this shape does not carry cannot be expressed, + // and silently dropping it would turn "force this value" into "whatever + // the caller sent". Fail loudly instead. + for _, name := range slices.Sorted(maps.Keys(shape.Defaults)) { + if _, ok := kept[name]; !ok { + return nil, nil, fmt.Errorf( + "cannot inject a default into column %q: the role's schema does not carry it", name) + } + } + return cols, wire, nil +} + +type roleEntry struct { + key string + table *Table // nil: this shape does not compile + cause string +} + +// roleCache is an LRU of per-role projections, with the same discipline as +// filterCache: bounded, negative entries cached, and every evicted handle +// closed. +type roleCache struct { + mu sync.Mutex + cap int + order *list.List // front = most recently used + index map[string]*list.Element +} + +func newRoleCache(capacity int) *roleCache { + return &roleCache{cap: capacity, order: list.New(), index: make(map[string]*list.Element)} +} + +// capacity is the cache's bound, for a rebind's fresh cache; roleCacheSize +// for a nil cache. +func (c *roleCache) capacity() int { + if c == nil { + return roleCacheSize + } + return c.cap +} + +// evictOldestLocked drops the least recently used entry and returns the handle +// the caller must close. Closing is the caller's job because it waits for that +// projection's readers, and waiting under the cache lock would stall every +// other role on the table. +func (c *roleCache) evictOldestLocked() *Table { + el := c.order.Back() + if el == nil { + return nil + } + c.order.Remove(el) + e := el.Value.(*roleEntry) + delete(c.index, e.key) + return e.table +} + +// closeAll drops every projection. Called once the cache is detached from its +// base table (see Table.detachLocked), so no new lookup can reach it; each +// projection's own write lock waits for the requests already holding it. +func (c *roleCache) closeAll() { + c.mu.Lock() + defer c.mu.Unlock() + for el := c.order.Front(); el != nil; el = el.Next() { + if rt := el.Value.(*roleEntry).table; rt != nil { + rt.close("base table rebound or closed") + } + } + c.order.Init() + c.index = make(map[string]*list.Element) +} diff --git a/internal/typelayer/roletable_test.go b/internal/typelayer/roletable_test.go new file mode 100644 index 00000000..7588b4ab --- /dev/null +++ b/internal/typelayer/roletable_test.go @@ -0,0 +1,301 @@ +package typelayer + +import ( + "encoding/json" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// ordersTable is the per-role fixture: an ordinary column a role may be denied +// (secret), one a check clause injects into (tenant), and a numeric column +// whose reader refuses a text literal (amount). +func ordersTable() *discovery.TableSchema { + return &discovery.TableSchema{ + Name: "orders", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt32", Position: 1}, + {Name: "tenant", Type: "String", Position: 2}, + {Name: "secret", Type: "String", Position: 3}, + {Name: "amount", Type: "UInt64", Position: 4}, + }, + } +} + +func roleTableFor(t *testing.T, eng *Engine, shape RoleShape) *Table { + t.Helper() + tbl, err := eng.RoleTable(tenant.Default, "orders", shape) + require.NoError(t, err) + t.Cleanup(tbl.Release) + return tbl +} + +// TestRoleTable_IdentityShapeIsTheBaseTable: a role that may write everything +// and injects nothing costs no second compile and no cache entry. +func TestRoleTable_IdentityShapeIsTheBaseTable(t *testing.T) { + eng := TestEngine(t, ordersTable()) + + base, err := eng.Table(tenant.Default, "orders") + require.NoError(t, err) + base.Release() + + role, err := eng.RoleTable(tenant.Default, "orders", RoleShape{}) + require.NoError(t, err) + defer role.Release() + + assert.Same(t, base, role) + assert.Equal(t, 0, base.roles.len()) +} + +// TestRoleTable_DeniedColumnIsAbsentFromTheSchema: a column the role may not +// write is simply not in the compiled DDL, so a record +// naming it is ClickHouse's own per-row code 117 rather than a Go key walk's +// 403 — and the exported row carries the ROLE's column list. +func TestRoleTable_DeniedColumnIsAbsentFromTheSchema(t *testing.T) { + eng := TestEngine(t, ordersTable()) + tbl := roleTableFor(t, eng, RoleShape{Columns: []string{"id", "tenant", "amount"}}) + + assert.Equal(t, []string{"id", "tenant", "amount"}, tbl.WireColumns) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte( + `{"id":1,"tenant":"acme","amount":5}`+"\n"+ + `{"id":2,"tenant":"acme","secret":"x","amount":5}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 2) + + assert.True(t, batch.Rows[0].Accepted) + assert.Equal(t, `[1, "acme", 5]`, string(batch.Rows[0].Line)) + + assert.False(t, batch.Rows[1].Accepted) + assert.False(t, batch.Rows[1].Declined, "a denied column is a verdict about the data, not a decline") + assert.Equal(t, 117, batch.Rows[1].Code) + assert.Contains(t, batch.Rows[1].Message, "secret") +} + +// TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue, measured on +// the 26.6 artifact: DEFAULT '' fills a column the record omits, and a +// value the record DOES supply still wins. +func TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue(t *testing.T) { + eng := TestEngine(t, ordersTable()) + tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"tenant": "acme"}}) + + assert.Equal(t, []string{"id", "tenant", "secret", "amount"}, tbl.WireColumns) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte( + `{"id":1,"amount":5}`+"\n"+ + `{"id":2,"tenant":"other","amount":5}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 2) + + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, `[1, "acme", "", 5]`, string(batch.Rows[0].Line)) + require.True(t, batch.Rows[1].Accepted, batch.Rows[1].Message) + assert.Equal(t, `[2, "other", "", 5]`, string(batch.Rows[1].Line)) +} + +// TestRoleTable_LiteralEscaping: the injected value is the one thing in this +// DDL that is SQL TEXT, spelled by the library's QuoteLiteral, so every +// spelling that could close the literal early has to survive as data. The +// breakout attempt is the case that matters — it must not add, remove or +// retype a single column. +func TestRoleTable_LiteralEscaping(t *testing.T) { + eng := TestEngine(t, ordersTable()) + + for name, value := range map[string]string{ + "apostrophe": "O'Brien", + "backslash": `back\slash`, + "backslash quote": `x\'y`, + "trailing escape": `ends with \`, + "ddl breakout": `', x UInt8 DEFAULT '`, + "comment breakout": `' --`, + "empty": "", + "nul": "a\x00b", + "newline and tab": "a\nb\tc", + "unicode": "Ünï 日本", + } { + t.Run(name, func(t *testing.T) { + tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"tenant": value}}) + + // An escape that closed the literal early would show up here, as an + // extra or missing column, not as a mangled value. + assert.Equal(t, []string{"id", "tenant", "secret", "amount"}, tbl.WireColumns) + assert.Equal(t, []string{"id", "tenant", "secret", "amount"}, + compiledColumnNames(tbl.pool.first().schema)) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":1,"amount":5}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + + // The exported line is ClickHouse's own writer on the stored value, + // so it is the ground truth for what the DEFAULT actually holds. + want, err := json.Marshal(value) + require.NoError(t, err) + assert.Equal(t, `[1, `+string(want)+`, "", 5]`, string(batch.Rows[0].Line)) + }) + } +} + +// TestRoleTable_UnparseableLiteralFailsClosed: a literal the column's reader +// cannot read is a compile refusal (ClickHouse code 6). It must be Unavailable +// — a 503 — never a handle that silently drops the injection. +func TestRoleTable_UnparseableLiteralFailsClosed(t *testing.T) { + eng := TestEngine(t, ordersTable()) + + _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Defaults: map[string]string{"amount": "abc"}}) + require.Error(t, err) + assert.True(t, IsUnavailable(err)) + assert.Contains(t, err.Error(), "orders") +} + +// TestRoleTable_ContradictoryShapeIsAnError: a default for a column the shape +// does not carry cannot be expressed. Dropping it silently would turn "force +// this value" into "whatever the caller sent". +func TestRoleTable_ContradictoryShapeIsAnError(t *testing.T) { + eng := TestEngine(t, ordersTable()) + + _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{ + Columns: []string{"id", "amount"}, + Defaults: map[string]string{"tenant": "acme"}, + }) + require.Error(t, err) + assert.Contains(t, err.Error(), "tenant") + + _, err = eng.RoleTable(tenant.Default, "orders", RoleShape{Defaults: map[string]string{"nosuch": "acme"}}) + require.Error(t, err) + assert.Contains(t, err.Error(), "nosuch") +} + +// TestRoleTable_DefaultIntoAComputedColumnIsRefused: MATERIALIZED/ALIAS/ +// EPHEMERAL are never supplied by a record, and rewriting one into a plain +// DEFAULT would change what the server stores. +func TestRoleTable_DefaultIntoAComputedColumnIsRefused(t *testing.T) { + ts := ordersTable() + ts.Columns = append(ts.Columns, discovery.Column{ + Name: "double", Type: "UInt64", HasDefault: true, + DefaultKind: "MATERIALIZED", DefaultExpression: "amount * 2", Position: 5, + }) + eng := TestEngine(t, ts) + + _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Defaults: map[string]string{"double": "1"}}) + require.Error(t, err) + assert.Contains(t, err.Error(), "MATERIALIZED") + + // A computed column is kept whatever the allow-list says: dropping it would + // change what the server computes. + tbl := roleTableFor(t, eng, RoleShape{Columns: []string{"id", "amount"}}) + assert.Equal(t, []string{"id", "amount"}, tbl.WireColumns) + assert.Equal(t, []string{"id", "amount", "double"}, compiledColumnNames(tbl.pool.first().schema)) +} + +// TestRoleTable_CachedPerShapeAndGeneration: the same shape must reuse the +// handle (a compile per request is the thing this cache exists to stop), a +// different shape must not, and a rebind must invalidate both. +func TestRoleTable_CachedPerShapeAndGeneration(t *testing.T) { + eng := TestEngine(t, ordersTable()) + + shape := RoleShape{Columns: []string{"tenant", "id", "amount"}, Defaults: map[string]string{"tenant": "acme"}} + role := func(s RoleShape) *Table { + tbl, err := eng.RoleTable(tenant.Default, "orders", s) + require.NoError(t, err) + // Released immediately: a rebind waits for every projection's readers, + // so a held handle would block Bind rather than be invalidated by it. + tbl.Release() + return tbl + } + + first := role(shape) + // Column ORDER is not part of the shape — the DDL always follows the + // table's declaration order. + assert.Same(t, first, role(RoleShape{ + Columns: []string{"amount", "tenant", "id"}, + Defaults: map[string]string{"tenant": "acme"}, + })) + assert.NotSame(t, first, role(RoleShape{ + Columns: []string{"tenant", "id", "amount"}, + Defaults: map[string]string{"tenant": "beta"}, + }), "the injected value is baked into the handle") + + base, err := eng.Table(tenant.Default, "orders") + require.NoError(t, err) + assert.Equal(t, 2, base.roles.len()) + base.Release() + + // A rebind closes every projection; the next lookup compiles a fresh one. + changed := ordersTable() + changed.Columns[3].Type = "UInt32" + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + + after := role(shape) + assert.NotSame(t, first, after) + assert.Equal(t, uint64(2), after.Generation) +} + +// TestRoleTable_NegativeEntryStopsRecompiling: a shape that will not compile +// costs one compile and one log line per generation, like filterCache. +func TestRoleTable_NegativeEntryStopsRecompiling(t *testing.T) { + eng := TestEngine(t, ordersTable()) + + for range 3 { + _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Defaults: map[string]string{"amount": "abc"}}) + require.Error(t, err) + } + base, err := eng.Table(tenant.Default, "orders") + require.NoError(t, err) + defer base.Release() + assert.Equal(t, 1, base.roles.len()) +} + +// TestRoleTable_BoundedUnderTenantValueChurn: the injected values come from +// tenant claims and are baked into compiled handles, so the cache must be +// bounded exactly like filterCache. +func TestRoleTable_BoundedUnderTenantValueChurn(t *testing.T) { + eng := TestEngine(t, ordersTable()) + + base, err := eng.Table(tenant.Default, "orders") + require.NoError(t, err) + base.Release() + base.mu.Lock() + base.roles = newRoleCache(4) + base.mu.Unlock() + + for i := range 20 { + tbl, err := eng.RoleTable(tenant.Default, "orders", RoleShape{ + Defaults: map[string]string{"tenant": string(rune('a' + i))}, + }) + require.NoError(t, err) + tbl.Release() + } + + base, err = eng.Table(tenant.Default, "orders") + require.NoError(t, err) + defer base.Release() + assert.Equal(t, 4, base.roles.len()) + + // The survivors still answer after their neighbours were closed. + tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"tenant": "acme"}}) + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":1,"amount":5}`+"\n")) + require.NoError(t, err) + require.True(t, batch.Rows[0].Accepted) + assert.Equal(t, `[1, "acme", "", 5]`, string(batch.Rows[0].Line)) +} + +// TestRoleTable_HasItsOwnHandlePool: a role shape holds its own single handle, +// not the base table's pool (256 shapes x the pool would be unbounded memory). +func TestRoleTable_HasItsOwnHandlePool(t *testing.T) { + eng := TestEngine(t, ordersTable()) + tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"tenant": "acme"}}) + assert.Len(t, tbl.pool.list(), roleHandles) + assert.Equal(t, int64(roleHandles), tbl.pool.limit.Load(), "a role shape never grows past its one handle") + assert.Nil(t, tbl.roles, "a projection is never itself projected") +} + +func (c *roleCache) len() int { + c.mu.Lock() + defer c.mu.Unlock() + return c.order.Len() +} diff --git a/internal/typelayer/tenancy_test.go b/internal/typelayer/tenancy_test.go new file mode 100644 index 00000000..dd76fcc4 --- /dev/null +++ b/internal/typelayer/tenancy_test.go @@ -0,0 +1,393 @@ +package typelayer + +import ( + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +const eventsRecord = `{"id":1,"name":"a","ts":"2026-01-15 10:30:00.000","tags":[],"score":null}` + "\n" + +// answers asserts that tenant id's events table compiles and judges a record. +func answers(t *testing.T, eng *Engine, id tenant.ID) { + t.Helper() + tbl, err := eng.Table(id, "events") + require.NoError(t, err, "tenant %s", id) + defer tbl.Release() + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(eventsRecord)) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + assert.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) +} + +// unavailable asserts that tenant id's events table answers *Unavailable for +// the tenant as a whole, and returns it. +func unavailable(t *testing.T, eng *Engine, id tenant.ID) *Unavailable { + t.Helper() + _, err := eng.Table(id, "events") + require.Error(t, err, "tenant %s", id) + var u *Unavailable + require.ErrorAs(t, err, &u) + assert.Equal(t, id, u.Tenant) + return u +} + +// TestBind_TenantsOnOneLineWithDifferentZones: the process opened the 26.6 +// line in UTC, so a second tenant on that line whose server reports another +// zone cannot be served by this process. It alone is refused, with a cause +// naming both zones; the first tenant, and a third in the line's own zone, +// keep answering. +func TestBind_TenantsOnOneLineWithDifferentZones(t *testing.T) { + eng := TestEngine(t, eventsTable()) // tenant.Default, UTC + + eng.Bind("berlin", TestServerVersion, "Europe/Berlin", []*discovery.TableSchema{eventsTable()}) + eng.Bind("utc", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind("unnamed", TestServerVersion, "", []*discovery.TableSchema{eventsTable()}) + + u := unavailable(t, eng, "berlin") + assert.Empty(t, u.Table, "the cause covers every table of the tenant") + assert.Contains(t, u.Cause, `"Europe/Berlin"`) + assert.Contains(t, u.Cause, `"UTC"`) + assert.Contains(t, u.Cause, "one timezone per ClickHouse version line") + assert.Contains(t, u.Error(), "berlin") + + answers(t, eng, tenant.Default) + answers(t, eng, "utc") + answers(t, eng, "unnamed") // "" is chtypes' own default, UTC + + // Role projections inherit the refusal: they are compiled off the base. + _, err := eng.RoleTable("berlin", "events", RoleShape{Columns: []string{"id"}}) + require.True(t, IsUnavailable(err)) +} + +// TestBind_ZoneRefusalClearsWhenTheZoneMatchesAgain: the refusal is about the +// zone the server reports now, not a sticky mark on the tenant. +func TestBind_ZoneRefusalClearsWhenTheZoneMatchesAgain(t *testing.T) { + eng := TestEngine(t, eventsTable()) + eng.Bind("moved", TestServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) + u := unavailable(t, eng, "moved") + assert.Contains(t, u.Cause, `"Asia/Tokyo"`) + + eng.Bind("moved", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + answers(t, eng, "moved") +} + +// TestBind_MissingLineIsThatTenantOnly: a tenant whose server is on a line no +// artifact covers is refused with the SDK's own message; every other tenant +// answers, and the tenant answers once its line resolves. +func TestBind_MissingLineIsThatTenantOnly(t *testing.T) { + eng := TestEngine(t, eventsTable()) + + eng.Bind("old", "1.2.3.4", "UTC", []*discovery.TableSchema{eventsTable()}) + u := unavailable(t, eng, "old") + assert.Contains(t, u.Cause, "Looked in:", "the SDK's message names every directory searched") + + answers(t, eng, tenant.Default) + + eng.Bind("old", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + answers(t, eng, "old") +} + +// TestTable_UnboundTenantIsUnavailable: a tenant discovery has not bound yet +// is a 503, never a 404 — the table may well exist. +func TestTable_UnboundTenantIsUnavailable(t *testing.T) { + eng := TestEngine(t, eventsTable()) + u := unavailable(t, eng, "nobody") + assert.Equal(t, causeUnbound, u.Cause) +} + +// TestBind_TenantsCompileSeparateHandles: two tenants with the same table do +// not share a handle (sharing is a separate decision), and one tenant's +// rebind leaves the other's generation alone. +func TestBind_TenantsCompileSeparateHandles(t *testing.T) { + eng := TestEngine(t, eventsTable()) + eng.Bind("other", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + + a, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + b, err := eng.Table("other", "events") + require.NoError(t, err) + assert.NotSame(t, a, b) + assert.NotSame(t, a.pool.first(), b.pool.first()) + a.Release() + b.Release() + + changed := eventsTable() + changed.Columns = append(changed.Columns, discovery.Column{Name: "extra", Type: "String", Position: 8}) + eng.Bind("other", TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + + a, err = eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer a.Release() + b, err = eng.Table("other", "events") + require.NoError(t, err) + defer b.Release() + assert.Equal(t, uint64(1), a.Generation) + assert.Equal(t, uint64(2), b.Generation) + assert.NotContains(t, a.WireColumns, "extra") + assert.Contains(t, b.WireColumns, "extra") +} + +// TestForget_ClosesOnlyThatTenantAndDoesNotBlock: Forget is called under the +// reload lock, which must never wait on a request. With one of the tenant's +// tables held, Forget still returns at once; the tenant is Unavailable from +// that moment; its handles close once the holder lets go; and another +// tenant's handles are untouched throughout. +func TestForget_ClosesOnlyThatTenantAndDoesNotBlock(t *testing.T) { + eng := TestEngine(t, eventsTable()) + eng.Bind("gone", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + + held, err := eng.Table("gone", "events") + require.NoError(t, err) + + returned := make(chan struct{}) + go func() { + eng.Forget("gone") + close(returned) + }() + select { + case <-returned: + case <-time.After(5 * time.Second): + t.Fatal("Forget blocked on a held table") + } + + u := unavailable(t, eng, "gone") + assert.Equal(t, causeUnbound, u.Cause) + answers(t, eng, tenant.Default) + + // The held handle still works: the teardown waits for it. + batch, err := held.Ingest(FormatJSONEachRow, []byte(eventsRecord)) + require.NoError(t, err) + assert.True(t, batch.Rows[0].Accepted) + + retired := make(chan struct{}) + go func() { + eng.retiring.Wait() + close(retired) + }() + select { + case <-retired: + t.Fatal("teardown finished while a request still held the table") + case <-time.After(50 * time.Millisecond): + } + held.Release() + select { + case <-retired: + case <-time.After(5 * time.Second): + t.Fatal("teardown did not finish after the holder released") + } + held.mu.RLock() + assert.Nil(t, held.pool, "the forgotten tenant's handles are closed") + assert.Equal(t, causeRetired, held.cause) + held.mu.RUnlock() + + answers(t, eng, tenant.Default) + + // A later Bind brings the tenant back on fresh handles. + eng.Bind("gone", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + answers(t, eng, "gone") +} + +// TestClose_WaitsForForgetAndRefusesAfterwards: shutdown leaves no teardown +// running, and nothing answers or binds after it. +func TestClose_WaitsForForgetAndRefusesAfterwards(t *testing.T) { + eng := TestEngine(t, eventsTable()) + eng.Bind("a", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Forget("a") + eng.Close() + + u := unavailable(t, eng, tenant.Default) + assert.Equal(t, causeClosed, u.Cause) + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + u = unavailable(t, eng, tenant.Default) + assert.Equal(t, causeClosed, u.Cause) + eng.Forget(tenant.Default) // no-op, no panic +} + +// TestPool_GrowsLazilyToItsLimit: one handle after Bind; a call that finds +// every handle busy compiles one more, up to the limit; past it, calls share +// a busy handle; and an idle handle is reused rather than grown past. +func TestPool_GrowsLazilyToItsLimit(t *testing.T) { + eng := TestEngine(t, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + defer tbl.Release() + + p, cause := newPool(tbl.lib, tbl.pool.ddl, 3) + require.Empty(t, cause) + t.Cleanup(p.close) + require.Len(t, p.list(), 1) + + a := p.acquire() + assert.Len(t, p.list(), 1, "an idle handle is taken, not grown past") + b := p.acquire() + assert.Len(t, p.list(), 2, "every handle busy: grow") + c := p.acquire() + assert.Len(t, p.list(), 3) + d := p.acquire() + assert.Len(t, p.list(), 3, "at the limit, a call shares a busy handle") + assert.NotSame(t, a, b) + assert.NotSame(t, b, c) + assert.Contains(t, []*schemaSlot{a, b, c}, d) + assert.Equal(t, int32(2), d.busy.Load()) + + for _, s := range []*schemaSlot{a, b, c, d} { + p.release(s) + } + for _, s := range p.list() { + assert.Zero(t, s.busy.Load()) + } + e := p.acquire() + assert.Len(t, p.list(), 3) + p.release(e) + for _, s := range p.list() { + assert.Equal(t, p.perSlot, s.filters.cap, "every grown slot gets the split filter budget") + } +} + +// TestTable_PoolGrowsUnderConcurrentHolders drives growth through the public +// path: a parsed Row keeps its handle busy until Close. A burst of concurrent +// parses grows the pool (a caller that loses the race to grow shares a busy +// handle rather than wait for a compile), never past its limit; holding Rows +// while more arrive grows it to exactly the limit. +func TestTable_PoolGrowsUnderConcurrentHolders(t *testing.T) { + eng := TestEngine(t, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + defer tbl.Release() + require.Len(t, tbl.pool.list(), 1, "one handle after Bind") + limit := int(tbl.pool.limit.Load()) + + parse := func() *Row { + row, err := tbl.ParseRow(tbl.WireColumns, []byte(sampleRow)) + require.NoError(t, err) + return row + } + + burst := make([]*Row, limit+2) + var wg sync.WaitGroup + for i := range burst { + wg.Go(func() { + row, err := tbl.ParseRow(tbl.WireColumns, []byte(sampleRow)) + if assert.NoError(t, err) { + burst[i] = row + } + }) + } + wg.Wait() + assert.LessOrEqual(t, len(tbl.pool.list()), limit, "never past the limit") + + held := append([]*Row(nil), burst...) + for range limit { + held = append(held, parse()) + } + assert.Len(t, tbl.pool.list(), limit, "sustained holders grow it to exactly the limit") + + // Every held Row still answers on its own handle. + for _, r := range held { + require.NotNil(t, r) + assert.True(t, r.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}})) + r.Close() + r.Close() // a second Close is a no-op, not a double release + } + for _, s := range tbl.pool.list() { + assert.Zero(t, s.busy.Load(), "every handle given back") + } +} + +// TestBind_HeldRoleTableDoesNotBlockBaseLookups: a rebind detaches the old +// role projections and closes them after releasing the base table, so a +// request that holds a projection and then looks the base table up is not +// deadlocked against the rebind waiting for that projection. +func TestBind_HeldRoleTableDoesNotBlockBaseLookups(t *testing.T) { + eng := TestEngine(t, ordersTable()) + rt, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Columns: []string{"id", "tenant"}}) + require.NoError(t, err) + + changed := ordersTable() + changed.Columns[3].Type = "UInt32" + bound := make(chan struct{}) + go func() { + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + close(bound) + }() + + require.Eventually(t, func() bool { + base, err := eng.Table(tenant.Default, "orders") + if err != nil { + return false + } + defer base.Release() + return base.Generation == 2 + }, 5*time.Second, 5*time.Millisecond, "the base table must be reachable while the old projection is held") + + select { + case <-bound: + t.Fatal("Bind returned before the held projection was released") + default: + } + rt.Release() + select { + case <-bound: + case <-time.After(5 * time.Second): + t.Fatal("Bind did not finish after the projection was released") + } +} + +// TestEngine_ConcurrentBindTableForget runs every tenancy entry point at once +// across a few tenants. Under -race it is the pin that the engine and tenant +// maps, the per-tenant bind serialization and the background teardown are +// safe; every answer is either a verdict or an Unavailable, never a panic or +// a hang. +func TestEngine_ConcurrentBindTableForget(t *testing.T) { + eng := TestEngine(t, eventsTable()) + ids := []tenant.ID{"t0", "t1", "t2", "t3"} + schemas := func(i int) []*discovery.TableSchema { + ts := eventsTable() + if i%2 == 1 { + ts.Columns = append(ts.Columns, discovery.Column{Name: "extra", Type: "String", Position: 8}) + } + return []*discovery.TableSchema{ts} + } + + var wg sync.WaitGroup + for g := range 12 { + wg.Go(func() { + for i := range 15 { + id := ids[(g+i)%len(ids)] + switch (g + i) % 4 { + case 0, 1: + eng.Bind(id, TestServerVersion, "UTC", schemas(i)) + case 2: + eng.Forget(id) + default: + tbl, err := eng.Table(id, "events") + if err != nil { + assert.True(t, IsUnavailable(err), "%v", err) + continue + } + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(eventsRecord)) + tbl.Release() + if assert.NoError(t, err) && assert.Len(t, batch.Rows, 1) { + assert.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + } + } + } + }) + } + wg.Wait() + eng.retiring.Wait() + + for _, id := range ids { + eng.Bind(id, TestServerVersion, "UTC", schemas(0)) + answers(t, eng, id) + } + answers(t, eng, tenant.Default) + assert.Len(t, eng.tenants, len(ids)+1) +} diff --git a/internal/typelayer/testing.go b/internal/typelayer/testing.go new file mode 100644 index 00000000..a626944c --- /dev/null +++ b/internal/typelayer/testing.go @@ -0,0 +1,94 @@ +package typelayer + +import ( + "fmt" + "os" + "slices" + "strings" + "sync" + "testing" + + "github.com/wave-rf/chtypes/go/chtypes" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// TestServerVersion is the ClickHouse version TestEngine binds to — a real +// 26.6 patch release, so the registry resolves the 26.6 artifact line. +const TestServerVersion = "26.6.3.62" + +// testLine is TestServerVersion's version line, the one artifact tests need. +const testLine = "26.6" + +// RequireEnv set to "1" turns every artifact skip into a failure. CI sets it, +// so a runner with a broken artifact cache fails loudly instead of quietly +// testing nothing. +const RequireEnv = "WAVEHOUSE_TEST_REQUIRE_CHTYPES" + +// missingArtifact is what a developer without the artifact needs to read: the +// exact command that installs it. +const missingArtifact = "chtypes artifact for " + testLine + " (ABI revision 6) not installed: run " + + "`scripts/fetch-chtypes.sh`, which installs the build chtypes.lock pins into the SDK's per-user cache" + +// SkipWithoutArtifact skips t when the chtypes artifact for TestServerVersion's +// line is not on the SDK's search path — or fails it when +// WAVEHOUSE_TEST_REQUIRE_CHTYPES=1. It opens no library. A test that boots +// anything constructing an Engine (the API process role) calls it first, since +// NewEngine refuses to start without an artifact. +func SkipWithoutArtifact(t testing.TB) { + t.Helper() + if cause := artifactMissing(); cause != "" { + skipUnlessRequired(t, cause) + } +} + +var artifactProbe struct { + once sync.Once + cause string +} + +// artifactMissing reports why the test line's artifact is not available, "" +// when it is. Probed once per test binary: it reads manifests only. +func artifactMissing() string { + artifactProbe.once.Do(func() { + reg, err := chtypes.NewRegistry("", chtypes.WithAutoFetch(false)) + if err != nil { + artifactProbe.cause = err.Error() + return + } + if !slices.Contains(reg.Versions(), testLine) { + artifactProbe.cause = fmt.Sprintf("no %s artifact on the registry search path (looked in: %s)", + testLine, strings.Join(reg.SearchPath(), ", ")) + } + }) + return artifactProbe.cause +} + +// TestEngine opens an Engine on the SDK's default search path and binds the +// given tables for tenant.Default at TestServerVersion in UTC. It skips the +// test when the artifact is absent or does not load (fails it under +// WAVEHOUSE_TEST_REQUIRE_CHTYPES=1), and closes the Engine when the test ends. +func TestEngine(t testing.TB, tables ...*discovery.TableSchema) *Engine { + t.Helper() + SkipWithoutArtifact(t) + eng, err := NewEngine(Config{}) + if err != nil { + skipUnlessRequired(t, err.Error()) + } + t.Cleanup(eng.Close) + + eng.Bind(tenant.Default, TestServerVersion, "UTC", tables) + if cause := eng.tenantCause(tenant.Default); cause != "" { + skipUnlessRequired(t, cause) + } + return eng +} + +func skipUnlessRequired(t testing.TB, cause string) { + t.Helper() + if os.Getenv(RequireEnv) == "1" { + t.Fatalf("%s\n%s", missingArtifact, cause) + } + t.Skipf("%s\n%s", missingArtifact, cause) +} diff --git a/internal/typelayer/typelayer.go b/internal/typelayer/typelayer.go new file mode 100644 index 00000000..8a468648 --- /dev/null +++ b/internal/typelayer/typelayer.go @@ -0,0 +1,596 @@ +// Package typelayer owns every call into the chtypes library. It compiles one +// ClickHouse schema handle per discovered table, per tenant, against the +// artifact for that tenant's server version line, and answers the questions +// the gateway would otherwise have to re-derive in Go, with the server's own +// parser: +// +// - "would this record insert, and does it satisfy a role's insert check +// clauses?" — Table.Ingest +// - "does this stored row match a role's row filter?" — Table.ParseRow / +// Row.Visible +// - "what would this record insert for a role that may not write every +// column, and whose absent check columns must be filled?" — +// Engine.RoleTable +// +// Nothing outside this package imports github.com/wave-rf/chtypes/go/chtypes. +package typelayer + +import ( + "fmt" + "log/slog" + "strings" + "sync" + + "github.com/wave-rf/chtypes/go/chtypes" + + "github.com/Wave-RF/WaveHouse/internal/chsql" + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// Format is the wire format a body is parsed as. It is chtypes' own enum, +// aliased so no other package has to import the SDK to name one; only the +// constants below are formats Ingest accepts. +type Format = chtypes.Format + +// The formats Ingest accepts. JSONEachRow is name-addressed (an NDJSON body is +// the same format, byte for byte); CSV and TSV are positional in declaration +// order, with ClickHouse's header auto-detection unless IngestOptions turns it +// off; the WithNames pair open with a header line that names the columns, in +// any order. +const ( + FormatJSONEachRow = chtypes.JSONEachRow + FormatCSV = chtypes.CSV + FormatTSV = chtypes.TSV + FormatCSVWithNames = chtypes.CSVWithNames + FormatTSVWithNames = chtypes.TSVWithNames +) + +// Causes an Unavailable carries for states of the Engine rather than of a +// table or an artifact. +const ( + causeUnbound = "no schema is bound for this tenant yet (discovery has not completed a refresh)" + causeRetired = "tenant retired" + causeDropped = "table no longer discovered" + causeClosed = "type layer closed" +) + +// Config is boot-tier: the registry directory is read once at process start. +// "" means the SDK's own search path ($CHTYPES_REGISTRY, the per-user cache, +// then the system directories); an explicit directory is searched first, then +// the rest of that path. Either way a library is opened lazily, by the first +// Bind for its line (~120 MB resident each). +type Config struct { + RegistryDir string +} + +// Engine is the process's one chtypes registry plus every tenant's compiled +// tables. It is opened once at boot; discovery rebinds a tenant after each of +// that tenant's successful refreshes (Bind), and the tenant's teardown drops +// it (Forget). +// +// Tenants are independent: a tenant whose server line has no artifact, whose +// server zone this process cannot adopt, or whose table does not compile is +// Unavailable on its own, and every other tenant keeps answering. Two tenants +// on the same server and database still compile separate handles. +type Engine struct { + reg *chtypes.Registry + + mu sync.RWMutex + tenants map[tenant.ID]*tenantSet + closed bool + // retiring tracks Forget's background teardowns, so Close (and a test) + // can wait for them. + retiring sync.WaitGroup +} + +// tenantSet is one tenant's compiled tables. +type tenantSet struct { + id tenant.ID + reg *chtypes.Registry + + // bindMu serializes Bind for this tenant: a manual refresh can overlap the + // auto-refresh loop, and two binds racing would leak a handle neither of + // them installed. retire takes it too, so a teardown waits for a bind in + // flight instead of racing it. + bindMu sync.Mutex + // retired is set, under bindMu, once the set is torn down; a Bind that + // finds it set starts over on a fresh set. + retired bool + + mu sync.RWMutex + // lib is the library the current handles were compiled against; a change + // invalidates every handle, because a filter or block only answers for the + // library its schema came from. + lib *chtypes.Library + // cause, when non-empty, is what every Table() of this tenant reports. + cause string + tables map[string]*Table +} + +// NewEngine opens the registry, which reads manifests and opens no library. It +// fails only when an explicit directory cannot be read or nothing on the search +// path holds an artifact, and returns the SDK's own message, which names the +// directories it looked in. Anything an artifact itself can be wrong about — a +// missing line, a refused ABI revision, a truncated library — surfaces at the +// first Bind for that line, as that tenant's Unavailable. +// +// WithPreload is deliberately not used: it opens a library at construction, +// and chtypes.Timezone must be set before that from a server's own zone, +// which only discovery knows (see openLine). +func NewEngine(cfg Config) (*Engine, error) { + reg, err := chtypes.NewRegistry(cfg.RegistryDir, chtypes.WithAutoFetch(false)) + if err != nil { + return nil, err + } + return &Engine{reg: reg, tenants: make(map[tenant.ID]*tenantSet)}, nil +} + +// Table returns tenant id's current compiled handle for a table, read-locked. +// The caller must Release it when done; the handle stays alive and stable for +// the whole time it is held, so a concurrent rebind waits rather than pulling +// the schema out from under a request. +// +// The lock is not reentrant: a caller holding a Table must not ask for the +// same table again (Table or RoleTable) before releasing it, or it deadlocks +// against a rebind waiting between the two. +func (e *Engine) Table(id tenant.ID, name string) (*Table, error) { + e.mu.RLock() + set, closed := e.tenants[id], e.closed + e.mu.RUnlock() + if closed { + return nil, &Unavailable{Tenant: id, Cause: causeClosed} + } + if set == nil { + return nil, &Unavailable{Tenant: id, Cause: causeUnbound} + } + + set.mu.RLock() + cause, t := set.cause, set.tables[name] + set.mu.RUnlock() + if cause != "" { + return nil, &Unavailable{Tenant: id, Cause: cause} + } + if t == nil { + return nil, &Unavailable{Tenant: id, Table: name, Cause: "no compiled schema (table not discovered)"} + } + t.mu.RLock() + if t.pool == nil { + cause := t.cause + t.mu.RUnlock() + return nil, &Unavailable{Tenant: id, Table: name, Cause: cause} + } + return t, nil +} + +// Bind resolves the library for the tenant's serverVersion and (re)compiles a +// handle per table, keeping every table whose columns and library are +// unchanged. It is called synchronously from discovery's refresh hook, so it +// must never be fatal: a failure is recorded as a cause — per table for a +// compile refusal, tenant-wide for a missing artifact or a zone mismatch — +// and surfaces as *Unavailable from Table. A table the tenant no longer has +// is closed; a tenant whose line resolves again is answering again. +// +// serverTZ is the server's default zone name as ClickHouse reports it; "" +// means UTC, chtypes' own default, never the host's zone. +func (e *Engine) Bind(id tenant.ID, serverVersion, serverTZ string, tables []*discovery.TableSchema) { + for { + set := e.tenantForBind(id) + if set == nil { + return // closed + } + set.bindMu.Lock() + if set.retired { + // Forgotten between the lookup and the lock: start over, on the + // fresh set a lookup now creates. + set.bindMu.Unlock() + continue + } + set.bind(serverVersion, serverTZ, tables) + set.bindMu.Unlock() + return + } +} + +// tenantForBind returns id's set, creating it, or nil once the engine is +// closed. +func (e *Engine) tenantForBind(id tenant.ID) *tenantSet { + e.mu.Lock() + defer e.mu.Unlock() + if e.closed { + return nil + } + set := e.tenants[id] + if set == nil { + set = &tenantSet{id: id, reg: e.reg, tables: make(map[string]*Table)} + e.tenants[id] = set + } + return set +} + +// Forget drops a tenant: every later Table for it is Unavailable until a Bind +// binds it afresh. Its handles are closed in the background — closing a +// handle waits for the requests holding it, and the caller may be holding a +// lock that must not wait on I/O — so Forget itself never blocks on them. +func (e *Engine) Forget(id tenant.ID) { + e.mu.Lock() + set := e.tenants[id] + delete(e.tenants, id) + if set != nil && !e.closed { + e.retiring.Add(1) + } + closed := e.closed + e.mu.Unlock() + if set == nil || closed { + return + } + go func() { + defer e.retiring.Done() + set.retire(causeRetired) + }() +} + +// Close releases every compiled handle, waiting for the requests holding them +// and for any background Forget. Libraries are never unloaded — chtypes +// deliberately has no dlclose path — so this is only for tests and shutdown; +// every later Table is Unavailable and every later Bind does nothing. +func (e *Engine) Close() { + e.mu.Lock() + sets := e.tenants + e.tenants = make(map[tenant.ID]*tenantSet) + e.closed = true + e.mu.Unlock() + for _, set := range sets { + set.retire(causeClosed) + } + e.retiring.Wait() +} + +// tenantCause is the tenant-wide cause Table would report, "" when the tenant +// is bound and answering, for TestEngine. +func (e *Engine) tenantCause(id tenant.ID) string { + e.mu.RLock() + set := e.tenants[id] + e.mu.RUnlock() + if set == nil { + return causeUnbound + } + set.mu.RLock() + defer set.mu.RUnlock() + return set.cause +} + +// retire tears the set down, after any bind in flight finishes. +func (s *tenantSet) retire(cause string) { + s.bindMu.Lock() + defer s.bindMu.Unlock() + s.retired = true + s.mu.Lock() + tables := s.tables + s.tables = make(map[string]*Table) + s.cause = cause + s.mu.Unlock() + for _, t := range tables { + t.close(cause) + } +} + +// bind is Bind's body, run under s.bindMu. +func (s *tenantSet) bind(serverVersion, serverTZ string, tables []*discovery.TableSchema) { + tz := serverTZ + if tz == "" { + tz = "UTC" + } + lib, cause := openLine(s.reg, serverVersion, tz) + if cause != "" { + // The tables stay: if the tenant's line resolves again in the same + // library, unchanged handles are kept rather than recompiled. + s.mu.Lock() + s.cause = cause + s.mu.Unlock() + slog.Error("chtypes cannot serve this tenant; its ingest and row filters are unavailable", + "tenant", s.id, "server_version", serverVersion, "server_tz", tz, "cause", cause) + return + } + + // Compile outside every lock: a handle costs milliseconds and Table() + // readers are on the request path. + type pending struct { + ts *discovery.TableSchema + sig string + pool *pool + cause string + wire []string + cols map[string]filterColumn + } + + s.mu.RLock() + current := make(map[string]*Table, len(s.tables)) + for k, v := range s.tables { + current[k] = v + } + libChanged := s.lib != lib + s.mu.RUnlock() + + fresh := make([]pending, 0, len(tables)) + keep := make(map[string]struct{}, len(tables)) + for _, ts := range tables { + keep[ts.Name] = struct{}{} + sig := signature(ts) + if !libChanged { + if old, exists := current[ts.Name]; exists && old.answers(sig) { + continue // same columns, same library: the handle still answers + } + } + p := pending{ts: ts, sig: sig} + p.pool, p.cause = compile(lib, ts) + if p.cause == "" { + schema := p.pool.first().schema + p.wire = deriveWireColumns(schema, wireColumns(ts)) + if p.cols, p.cause = declaredColumns(lib, schema, ts.Columns); p.cause != "" { + p.pool.close() + p.pool = nil + } + } + if p.cause != "" { + slog.Error("chtypes could not compile table schema", "tenant", s.id, "table", ts.Name, "cause", p.cause) + } + fresh = append(fresh, p) + } + + // Swap each table under its own lock and NOT under s.mu: taking a table's + // write lock waits for that table's in-flight requests, and holding the + // tenant lock through that wait would stall lookups for every other table. + added := make(map[string]*Table, len(fresh)) + for _, p := range fresh { + t, exists := current[p.ts.Name] + if !exists { + // Nobody holds a pointer to a new table yet, so it is filled before + // it is published rather than swapped. + t = &Table{Name: p.ts.Name, tenant: s.id, Generation: 1, roles: newRoleCache(roleCacheSize)} + t.install(p.pool, p.cause, p.wire, p.cols, p.sig, lib, p.ts.Columns) + added[p.ts.Name] = t + continue + } + t.mu.Lock() + old := t.detachLocked() + t.Generation++ + t.roles = newRoleCache(old.roles.capacity()) + t.install(p.pool, p.cause, p.wire, p.cols, p.sig, lib, p.ts.Columns) + t.mu.Unlock() + old.close() + } + + s.mu.Lock() + s.lib = lib + s.cause = "" + for name, t := range added { + s.tables[name] = t + } + var dropped []*Table + for name, t := range s.tables { + if _, still := keep[name]; !still { + dropped = append(dropped, t) + delete(s.tables, name) + } + } + s.mu.Unlock() + + for _, t := range dropped { + t.close(causeDropped) + } +} + +// Table is one compiled shape: a pool of identical schema handles plus the +// caches built over them. Fields are written only under the exclusive lock a +// rebind takes, so a holder of a Release-pending read lock sees a consistent +// set. +// +// A Table is either the discovered table's own schema (Engine.Table) or a +// per-role projection of it (Engine.RoleTable). The two have the same method +// set; a role Table's WireColumns are the ROLE's columns, which is what makes +// the exported row per-role. +type Table struct { + Name string + Generation uint64 + // WireColumns is declaration order minus MATERIALIZED/ALIAS/EPHEMERAL — + // exactly the columns RowsExport emits, and therefore exactly what the + // message envelope's Columns must carry. It is read off the COMPILED handle + // (LoadedSchema.Columns), not recomputed from discovery, so it is a + // property of the thing that produced the bytes. + WireColumns []string + + tenant tenant.ID + mu sync.RWMutex + // pool holds the identically-compiled handles; nil means unavailable. + pool *pool + // cols maps every column the compiled schema declares, of every kind, to + // how render writes a predicate over it — what render tests a predicate's + // column against. + cols map[string]filterColumn + cause string // why pool is nil + sig string + // lib and discovered are what a per-role recompile needs: the library the + // handles came from, and the column list the DDL was built from. + lib *chtypes.Library + discovered []discovery.Column + // roles caches per-role projections of this table; nil on a role Table, + // which is never itself projected. + roles *roleCache +} + +// Release drops the read lock taken by Engine.Table or Engine.RoleTable. +func (t *Table) Release() { t.mu.RUnlock() } + +// answers reports whether the table already serves sig, compiled. +func (t *Table) answers(sig string) bool { + t.mu.RLock() + defer t.mu.RUnlock() + return t.sig == sig && t.pool != nil +} + +// install sets a freshly compiled shape. The caller holds t.mu exclusively, +// or is the only goroutine that can see t. +func (t *Table) install(p *pool, cause string, wire []string, cols map[string]filterColumn, sig string, + lib *chtypes.Library, discovered []discovery.Column, +) { + t.pool, t.cause = p, cause + t.WireColumns, t.cols, t.sig = wire, cols, sig + t.lib, t.discovered = lib, discovered +} + +// detached is what a table held before a swap or a close: closed once the +// table's own lock is released, so its holders never wait behind it. +type detached struct { + pool *pool + roles *roleCache +} + +// detachLocked takes the handles and the role projections off the table. The +// caller holds t.mu exclusively — so no request is using a slot of the +// detached pool — and closes the result after unlocking. +func (t *Table) detachLocked() detached { + d := detached{pool: t.pool, roles: t.roles} + t.pool, t.roles = nil, nil + return d +} + +// close releases what d held. A role projection waits for its own readers. +func (d detached) close() { + if d.roles != nil { + d.roles.closeAll() + } + d.pool.close() +} + +// close tears every handle down, waiting for this shape's readers first; a +// later Table() reports cause. +func (t *Table) close(cause string) { + t.mu.Lock() + d := t.detachLocked() + t.cause = cause + t.mu.Unlock() + d.close() +} + +// compile reconstructs the column-declaration list chtypes wants (not a CREATE +// TABLE) and compiles the table's first handle; the rest of its pool is +// compiled on contention. The engine and TTL clauses are deliberately not +// declared: chtypes declines engines it cannot model, and neither affects the +// insert verdicts or filter semantics this package asks for. +func compile(lib *chtypes.Library, ts *discovery.TableSchema) (*pool, string) { + cols := make([]chtypes.DiscoveredColumn, 0, len(ts.Columns)) + for _, c := range ts.Columns { + cols = append(cols, chtypes.DiscoveredColumn{ + Name: c.Name, + Type: c.Type, + DefaultKind: c.DefaultKind, + DefaultExpression: c.DefaultExpression, + Position: c.Position, + }) + } + ddl, err := lib.ReconstructDDL(cols) + if err != nil { + return nil, "cannot reconstruct column declarations: " + err.Error() + } + return newPool(lib, ddl, poolSize()) +} + +// signature is the column shape a handle was compiled from. An unchanged +// signature keeps the handle and the generation, so a refresh that discovers +// nothing new costs no compiles and invalidates no cached filter. +func signature(ts *discovery.TableSchema) string { + var b strings.Builder + for _, c := range ts.Columns { + fmt.Fprintf(&b, "%d\x1f%s\x1f%s\x1f%s\x1f%s\x1e", + c.Position, c.Name, c.Type, c.DefaultKind, c.DefaultExpression) + } + return b.String() +} + +// deriveWireColumns is what RowsExport emits, read off the compiled handle: +// declaration order minus the three kinds a positional INSERT never carries. +// MATERIALIZED and ALIAS are computed by the server; EPHEMERAL is insert-only +// and never stored. +// +// Taking it from LoadedSchema.Columns rather than recomputing it from +// discovery makes the wire list a property of the handle that produced the +// bytes, which is what ParseRow's drift check actually wants. An artifact +// without column introspection reports no Columns at all; the fallback is then +// the discovery-side computation, which answered identically on every +// artifact measured. +func deriveWireColumns(schema *chtypes.LoadedSchema, fallback []string) []string { + if schema == nil || len(schema.Columns) == 0 { + return fallback + } + out := make([]string, 0, len(schema.Columns)) + for _, c := range schema.Columns { + switch c.DefaultKind { + case chtypes.KindMaterialized, chtypes.KindAlias, chtypes.KindEphemeral: + // Never serialized by RowsExport. + case chtypes.KindNone, chtypes.KindDefault: + out = append(out, c.Name) + default: + // A kind this build does not know is treated as ordinary, matching + // the discovery-side fallback: dropping it would silently change + // the arity of every exported line. + out = append(out, c.Name) + } + } + return out +} + +// wireColumns is deriveWireColumns' discovery-side fallback. +func wireColumns(ts *discovery.TableSchema) []string { + out := make([]string, 0, len(ts.Columns)) + for _, c := range ts.Columns { + switch c.DefaultKind { + case "MATERIALIZED", "ALIAS", "EPHEMERAL": + default: + out = append(out, c.Name) + } + } + return out +} + +// filterColumn is how render writes a predicate over one column. +type filterColumn struct { + ident string // lib.QuoteIdentifier's spelling + intType chsql.IntType // "" unless the claim binds through chsql.StrictInt +} + +// declaredColumns maps every column name the compiled schema knows, of every +// kind, to its identifier as lib.QuoteIdentifier spells it (ClickHouse's own +// backQuote, always quoted) and, for an integer column, the type its claims +// are strictly cast to. It is the set render tests a predicate's column +// against: answering "no such column" here keeps a misspelled policy from +// costing a compile and a log line per generation, and on a ROLE table it is +// what makes a filter over a denied column fail closed instead of compiling +// against a column that is not there. Quoting once per compile keeps a C call +// off render's per-event path. A non-empty second return is the cause. +func declaredColumns(lib *chtypes.Library, schema *chtypes.LoadedSchema, fallback []discovery.Column) (map[string]filterColumn, string) { + type named struct{ name, typ string } + var cols []named + if schema != nil && len(schema.Columns) > 0 { + for _, c := range schema.Columns { + cols = append(cols, named{c.Name, c.Type}) + } + } else { + for _, c := range fallback { + cols = append(cols, named{c.Name, c.Type}) + } + } + m := make(map[string]filterColumn, len(cols)) + for _, c := range cols { + q, err := lib.QuoteIdentifier(c.name) + if err != nil { + return nil, fmt.Sprintf("cannot quote column %q: %s", c.name, err) + } + col := filterColumn{ident: q} + if it, ok := chsql.IntegerType(c.typ); ok { + col.intType = it + } + m[c.name] = col + } + return m, "" +} diff --git a/internal/typelayer/typelayer_test.go b/internal/typelayer/typelayer_test.go new file mode 100644 index 00000000..f039b05e --- /dev/null +++ b/internal/typelayer/typelayer_test.go @@ -0,0 +1,429 @@ +package typelayer + +import ( + "fmt" + "os" + "path/filepath" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/wave-rf/chtypes/go/chtypes" + + "github.com/Wave-RF/WaveHouse/internal/chsql" + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// eventsTable is the measured fixture: every column kind the ingest path has +// to answer for, including the one that never reaches the wire. +func eventsTable() *discovery.TableSchema { + return &discovery.TableSchema{ + Name: "events", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt8", Position: 1}, + {Name: "name", Type: "String", Position: 2}, + {Name: "ts", Type: "DateTime64(3)", Position: 3}, + {Name: "tags", Type: "Array(String)", Position: 4}, + {Name: "score", Type: "Nullable(Int32)", IsNullable: true, Position: 5}, + {Name: "created", Type: "DateTime", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "now()", Position: 6}, + {Name: "id_plus", Type: "UInt16", HasDefault: true, DefaultKind: "MATERIALIZED", DefaultExpression: "id + 1", Position: 7}, + }, + } +} + +func TestBind_CompilesAndExposesWireColumns(t *testing.T) { + eng := TestEngine(t, eventsTable()) + + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer tbl.Release() + + assert.Equal(t, uint64(1), tbl.Generation) + // MATERIALIZED never crosses the wire: RowsExport does not serialize it and + // the envelope's column list must match the exported line's arity. + assert.Equal(t, []string{"id", "name", "ts", "tags", "score", "created"}, tbl.WireColumns) +} + +func TestTable_UnknownTableIsUnavailable(t *testing.T) { + eng := TestEngine(t, eventsTable()) + _, err := eng.Table(tenant.Default, "nosuch") + require.Error(t, err) + assert.True(t, IsUnavailable(err)) + assert.Contains(t, err.Error(), "nosuch") +} + +// TestBind_GenerationBumpsOnlyOnSignatureChange: a refresh that discovers the +// same columns must not invalidate handles or cached filters, and one that +// discovers a new column must. +func TestBind_GenerationBumpsOnlyOnSignatureChange(t *testing.T) { + eng := TestEngine(t, eventsTable()) + + generation := func() uint64 { + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer tbl.Release() + return tbl.Generation + } + require.Equal(t, uint64(1), generation()) + + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + assert.Equal(t, uint64(1), generation(), "identical columns keep the handle") + + changed := eventsTable() + changed.Columns = append(changed.Columns, discovery.Column{Name: "extra", Type: "String", Position: 8}) + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer tbl.Release() + assert.Equal(t, uint64(2), tbl.Generation) + assert.Contains(t, tbl.WireColumns, "extra") +} + +// TestBind_DroppedTableBecomesUnavailable: a table that leaves the database +// must stop answering rather than serve a handle for a schema that is gone. +func TestBind_DroppedTableBecomesUnavailable(t *testing.T) { + eng := TestEngine(t, eventsTable()) + eng.Bind(tenant.Default, TestServerVersion, "UTC", nil) + + _, err := eng.Table(tenant.Default, "events") + require.Error(t, err) + assert.True(t, IsUnavailable(err)) +} + +// TestBind_MissingArtifact_UnavailableWithSDKMessage: the SDK's own text names +// every directory it searched — that is the whole diagnostic, so it is passed +// through verbatim. +func TestBind_MissingArtifact_UnavailableWithSDKMessage(t *testing.T) { + eng := TestEngine(t, eventsTable()) + + eng.Bind(tenant.Default, "1.2.3.4", "UTC", []*discovery.TableSchema{eventsTable()}) + _, err := eng.Table(tenant.Default, "events") + require.Error(t, err) + require.True(t, IsUnavailable(err)) + assert.Contains(t, err.Error(), "no artifact for ClickHouse 1.2") + assert.Contains(t, err.Error(), "Looked in:") + + // Rebinding a version that does resolve clears the tenant's cause. + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + tbl.Release() +} + +// TestBind_TimezoneMismatchIsTenantWideAndNamesBothZones: chtypes reads its +// Timezone global once per library open, so a server that changes zone cannot +// be adopted in-process. Every table of that tenant must stop answering, +// loudly. +func TestBind_TimezoneMismatchIsTenantWideAndNamesBothZones(t *testing.T) { + eng := TestEngine(t, eventsTable()) + + eng.Bind(tenant.Default, TestServerVersion, "Europe/Berlin", []*discovery.TableSchema{eventsTable()}) + _, err := eng.Table(tenant.Default, "events") + require.Error(t, err) + require.True(t, IsUnavailable(err)) + assert.Contains(t, err.Error(), "Europe/Berlin") + assert.Contains(t, err.Error(), "UTC") + assert.Contains(t, err.Error(), "restart") +} + +// TestBind_CompileRefusalIsPerTable: one undeclarable table must not take the +// rest of the database down with it. +func TestBind_CompileRefusalIsPerTable(t *testing.T) { + broken := &discovery.TableSchema{ + Name: "broken", + Columns: []discovery.Column{{Name: "x", Type: "NotAType(9)", Position: 1}}, + } + eng := TestEngine(t, eventsTable(), broken) + + good, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + good.Release() + + _, err = eng.Table(tenant.Default, "broken") + require.Error(t, err) + require.True(t, IsUnavailable(err)) + assert.Contains(t, err.Error(), "broken") +} + +func TestSignatureDistinguishesEveryField(t *testing.T) { + t.Parallel() + base := &discovery.TableSchema{Columns: []discovery.Column{ + {Name: "a", Type: "UInt8", DefaultKind: "DEFAULT", DefaultExpression: "1", Position: 1}, + }} + sig := signature(base) + for _, mutate := range []func(c *discovery.Column){ + func(c *discovery.Column) { c.Name = "b" }, + func(c *discovery.Column) { c.Type = "UInt16" }, + func(c *discovery.Column) { c.DefaultKind = "MATERIALIZED" }, + func(c *discovery.Column) { c.DefaultExpression = "2" }, + func(c *discovery.Column) { c.Position = 2 }, + } { + other := &discovery.TableSchema{Columns: []discovery.Column{base.Columns[0]}} + mutate(&other.Columns[0]) + assert.NotEqual(t, sig, signature(other)) + } +} + +// renderExpr is what a test asserts against: the expression typelayer hands +// chtypes, so a change in quoting or parameter naming is visible. +func renderExpr(t *testing.T, tbl *Table, preds ...Predicate) (string, map[string]string) { + t.Helper() + expr, params, ok := tbl.render(preds) + require.True(t, ok) + return expr, params +} + +func TestRender_QuotesIdentifiersAndBindsEveryValue(t *testing.T) { + eng := TestEngine(t, eventsTable()) + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer tbl.Release() + + expr, params := renderExpr(t, tbl, + Predicate{Column: "name", Op: "=", Values: []string{"acme"}}, + Predicate{Column: "id", Op: "in", Values: []string{"1", "7"}}, + ) + // Every identifier is backticked by the library's own QuoteIdentifier, + // which always quotes. Every value is a bound {pN:String} parameter, never + // text; on an integer column (id is UInt8 here) each one is compared + // through the strict cast, element by element. + assert.Equal(t, "`name` = {p0:String} AND `id` IN ("+ + chsql.StrictInt("p1", "UInt8")+", "+chsql.StrictInt("p2", "UInt8")+")", expr) + assert.Equal(t, map[string]string{"p0": "acme", "p1": "1", "p2": "7"}, params) + + hostile := Predicate{Column: "name", Op: "=", Values: []string{"' OR 1=1 --"}} + expr, params = renderExpr(t, tbl, hostile) + assert.Equal(t, "`name` = {p0:String}", expr) + assert.Equal(t, "' OR 1=1 --", params["p0"]) + assert.NotContains(t, expr, "OR 1=1") +} + +// TestRender_IntegerColumnsBindThroughTheStrictCast: every operator on an +// integer column (Nullable included, cast to the bare type) renders the same +// chsql.StrictInt expression the query path renders; every other column keeps +// the plain {pN:String} form. +func TestRender_IntegerColumnsBindThroughTheStrictCast(t *testing.T) { + eng := TestEngine(t, eventsTable()) + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer tbl.Release() + + for _, op := range []string{"=", "!=", ">", "<"} { + expr, params := renderExpr(t, tbl, Predicate{Column: "score", Op: op, Values: []string{"-4"}}) + assert.Equal(t, "`score` "+op+" "+chsql.StrictInt("p0", "Int32"), expr, "Nullable(Int32) %s", op) + assert.Equal(t, map[string]string{"p0": "-4"}, params) + + expr, _ = renderExpr(t, tbl, Predicate{Column: "id", Op: op, Values: []string{"7"}}) + assert.Equal(t, "`id` "+op+" "+chsql.StrictInt("p0", "UInt8"), expr, "UInt8 %s", op) + + for _, col := range []string{"name", "ts", "created"} { + expr, _ = renderExpr(t, tbl, Predicate{Column: col, Op: op, Values: []string{"x"}}) + assert.Equal(t, "`"+col+"` "+op+" {p0:String}", expr, "%s %s", col, op) + } + } + expr, params := renderExpr(t, tbl, Predicate{Column: "score", Op: "in", Values: []string{"1", "2", "3"}}) + assert.Equal(t, "`score` IN ("+chsql.StrictInt("p0", "Int32")+", "+chsql.StrictInt("p1", "Int32")+", "+ + chsql.StrictInt("p2", "Int32")+")", expr) + assert.Equal(t, map[string]string{"p0": "1", "p1": "2", "p2": "3"}, params) + + // The MATERIALIZED UInt16 column is declared too, so a filter over it is + // rendered the same way. + expr, _ = renderExpr(t, tbl, Predicate{Column: "id_plus", Op: "=", Values: []string{"8"}}) + assert.Equal(t, "`id_plus` = "+chsql.StrictInt("p0", "UInt16"), expr) +} + +// TestRender_QuotesEveryIdentifier: a column whose name is a reserved word is +// a syntax error unquoted, so the always-quoting spelling is the one render +// uses. +func TestRender_QuotesEveryIdentifier(t *testing.T) { + reserved := &discovery.TableSchema{ + Name: "reserved", + Columns: []discovery.Column{{Name: "all", Type: "String", Position: 1}}, + } + eng := TestEngine(t, reserved) + tbl, err := eng.Table(tenant.Default, "reserved") + require.NoError(t, err) + defer tbl.Release() + + expr, _ := renderExpr(t, tbl, Predicate{Column: "all", Op: "=", Values: []string{"x"}}) + assert.Equal(t, "`all` = {p0:String}", expr) +} + +func TestRender_BackticksAnIdentifierThatNeedsIt(t *testing.T) { + odd := &discovery.TableSchema{ + Name: "odd", + Columns: []discovery.Column{{Name: "weird name", Type: "String", Position: 1}}, + } + eng := TestEngine(t, odd) + tbl, err := eng.Table(tenant.Default, "odd") + require.NoError(t, err) + defer tbl.Release() + + expr, _ := renderExpr(t, tbl, Predicate{Column: "weird name", Op: "=", Values: []string{"x"}}) + assert.Equal(t, "`weird name` = {p0:String}", expr) +} + +func TestRender_RefusesWhatItCannotExpress(t *testing.T) { + eng := TestEngine(t, eventsTable()) + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer tbl.Release() + + for name, pred := range map[string]Predicate{ + "unknown column": {Column: "nosuch", Op: "=", Values: []string{"x"}}, + "empty values": {Column: "name", Op: "=", Values: nil}, + "unknown operator": {Column: "name", Op: "like", Values: []string{"x"}}, + "multi-value equal": {Column: "name", Op: "=", Values: []string{"a", "b"}}, + } { + _, _, ok := tbl.render([]Predicate{pred}) + assert.False(t, ok, name) + } +} + +func TestFilterCache_EvictsAndClosesOldest(t *testing.T) { + t.Parallel() + c := newFilterCache(2) + for i := range 3 { + key := fmt.Sprintf("k%d", i) + c.index[key] = c.order.PushFront(&filterEntry{key: key}) + if c.order.Len() > c.cap { + c.evictOldestLocked() + } + } + assert.Equal(t, 2, c.order.Len()) + assert.NotContains(t, c.index, "k0") +} + +// TestBind_WireColumnsComeFromTheCompiledHandle: the wire list is read off +// LoadedSchema.Columns, so it is a property of the handle that produced the +// bytes rather than a second derivation from discovery that could drift from +// it. All four default kinds are present so the filter is exercised whole. +func TestBind_WireColumnsComeFromTheCompiledHandle(t *testing.T) { + kinds := &discovery.TableSchema{ + Name: "kinds", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt32", Position: 1}, + {Name: "plain", Type: "String", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "'zzz'", Position: 2}, + {Name: "mat", Type: "UInt32", HasDefault: true, DefaultKind: "MATERIALIZED", DefaultExpression: "id + 1", Position: 3}, + {Name: "ali", Type: "UInt32", HasDefault: true, DefaultKind: "ALIAS", DefaultExpression: "id + 2", Position: 4}, + {Name: "eph", Type: "UInt8", DefaultKind: "EPHEMERAL", Position: 5}, + }, + } + eng := TestEngine(t, kinds) + tbl, err := eng.Table(tenant.Default, "kinds") + require.NoError(t, err) + defer tbl.Release() + + assert.Equal(t, []string{"id", "plain"}, tbl.WireColumns) + assert.Equal(t, []string{"id", "plain", "mat", "ali", "eph"}, compiledColumnNames(tbl.pool.first().schema), + "every declared column is known to the handle, of every kind") + // render tests against the handle's own column set, not the wire list: a + // filter may name a MATERIALIZED column. + _, _, ok := tbl.render([]Predicate{{Column: "mat", Op: "=", Values: []string{"2"}}}) + assert.True(t, ok) + _, _, ok = tbl.render([]Predicate{{Column: "nosuch", Op: "=", Values: []string{"2"}}}) + assert.False(t, ok) +} + +// TestBind_HandlePoolPerTable: every table gets a pool of one handle, and a +// rebind replaces all of it. +func TestBind_HandlePoolPerTable(t *testing.T) { + eng := TestEngine(t, eventsTable()) + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + assert.Len(t, tbl.pool.list(), 1) + first := tbl.pool.first() + tbl.Release() + + changed := eventsTable() + changed.Columns[0].Type = "UInt16" + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + + tbl, err = eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer tbl.Release() + assert.Len(t, tbl.pool.list(), 1) + assert.NotSame(t, first, tbl.pool.first()) +} + +// compiledColumnNames is the column list a handle compiled to, in declaration +// order. +func compiledColumnNames(schema *chtypes.LoadedSchema) []string { + out := make([]string, 0, len(schema.Columns)) + for _, c := range schema.Columns { + out = append(out, c.Name) + } + return out +} + +// TestQuoteIdentifier_AgreesWithChsql: the stream (render, through the +// library's QuoteIdentifier) and the SQL path (chsql.QuoteIdent, which has no +// library to ask) must spell every name byte for byte alike, so a divergence +// fails here rather than in a customer's query. +func TestQuoteIdentifier_AgreesWithChsql(t *testing.T) { + eng := TestEngine(t, eventsTable()) + lib := boundLib(eng, tenant.Default) + require.NotNil(t, lib) + + corpus := []string{ + "x", "a`b", `a\b`, "`", `\`, "\\`", "a\\`b", "it's", `"q"`, "", "null", "NULL", "all", + "select", "from", "where", "weird name", "n.a", "Ünï", "日本", "\xff\xfe", "1abc", "?", "--", "/*", + "a\x00b", "\x00", "\\\n", "\n\\", + } + for c := range 256 { + corpus = append(corpus, string([]byte{byte(c)}), "a"+string([]byte{byte(c)})+"b") + } + for _, name := range corpus { + ours, err := lib.QuoteIdentifier(name) + require.NoError(t, err, "%q", name) + assert.Equal(t, ours, chsql.QuoteIdent(name), "name %q", name) + } +} + +// TestNewEngine_OpensNoLibraryAtConstruction: the registry is lazy, so an +// artifact that cannot load is not a boot failure; the first Bind for its line +// reports the SDK's own error as that tenant's Unavailable. +func TestNewEngine_OpensNoLibraryAtConstruction(t *testing.T) { + TestEngine(t) // skips (or fails under WAVEHOUSE_TEST_REQUIRE_CHTYPES) without the real artifact + + dir := t.TempDir() + line := filepath.Join(dir, "26.6") + require.NoError(t, os.MkdirAll(line, 0o750)) + require.NoError(t, os.WriteFile(filepath.Join(line, "manifest.json"), + []byte(`{"library":"libchtypes.so","clickhouse_version":"26.6.8.7-stable","clickhouse_minor":"26.6"}`), 0o600)) + + eng, err := NewEngine(Config{RegistryDir: dir}) + require.NoError(t, err, "a broken artifact is not a construction error") + t.Cleanup(eng.Close) + + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + _, err = eng.Table(tenant.Default, "events") + require.Error(t, err) + require.True(t, IsUnavailable(err)) + assert.Contains(t, err.Error(), line, "the SDK's message names the directory that failed") +} + +// TestNewEngine_UnreadableDirectoryFailsAtConstruction: a directory somebody +// named and that does not exist is a typo, reported at boot. +func TestNewEngine_UnreadableDirectoryFailsAtConstruction(t *testing.T) { + t.Parallel() + _, err := NewEngine(Config{RegistryDir: filepath.Join(t.TempDir(), "nosuch")}) + require.Error(t, err) + assert.Contains(t, err.Error(), "nosuch") +} + +// boundLib is the library a tenant's handles were compiled against. +func boundLib(eng *Engine, id tenant.ID) *chtypes.Library { + eng.mu.RLock() + set := eng.tenants[id] + eng.mu.RUnlock() + if set == nil { + return nil + } + set.mu.RLock() + defer set.mu.RUnlock() + return set.lib +} diff --git a/internal/typelayer/zone.go b/internal/typelayer/zone.go new file mode 100644 index 00000000..8372d692 --- /dev/null +++ b/internal/typelayer/zone.go @@ -0,0 +1,52 @@ +package typelayer + +import ( + "fmt" + "sync" + + "github.com/wave-rf/chtypes/go/chtypes" +) + +// This file is the one place the server-zone rule lives. +// +// chtypes reads its Timezone package global once per library, when the library +// is first opened, and a library is opened at most once per process (the SDK +// dedupes by path). So a zone is a fact about this process and one ClickHouse +// version line, not about an Engine or a tenant: the first tenant to open a +// line fixes its zone, and a tenant on that line whose server reports another +// zone cannot be served by this process. Every library open goes through +// openLine, so the record below is complete. + +// lineZones is the zone each opened library was initialised with, keyed by the +// SDK's own *Library (one per opened artifact, shared by every Registry). +var lineZones = struct { + mu sync.Mutex + zone map[*chtypes.Library]string +}{zone: map[*chtypes.Library]string{}} + +// openLine resolves the library for serverVersion, opening it in tz when this +// process has not opened it yet. A non-empty cause is why the tenant cannot be +// served: no loadable artifact for the line (the SDK's own message), or a line +// already opened in another zone. +func openLine(reg *chtypes.Registry, serverVersion, tz string) (*chtypes.Library, string) { + lineZones.mu.Lock() + defer lineZones.mu.Unlock() + + chtypes.Timezone = tz // read only if this For is what opens the library + lib, err := reg.For(chtypes.Version(serverVersion)) + if err != nil { + return nil, err.Error() + } + opened, seen := lineZones.zone[lib] + if !seen { + lineZones.zone[lib] = tz + return lib, "" + } + if opened != tz { + return nil, fmt.Sprintf( + "ClickHouse reports server timezone %q, but this process opened the chtypes library for ClickHouse %s in %q; "+ + "one process serves one timezone per ClickHouse version line, so serve this tenant from another process or restart", + tz, lib.Minor, opened) + } + return lib, "" +} From 499533163b9b5b82608f6a4cbc73e84a226d8289 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:47:10 -0400 Subject: [PATCH 06/70] feat(api)!: serve /v1/query and pipes through ClickHouse's HTTP interface The structured query and named pipes now read through ClickHouse's HTTP interface and serve ClickHouse's own FORMAT JSONEachRow rendering, framed as a JSON array, instead of scanning rows through the native driver and re-marshalling them in Go. - Per-tenant Target and the TLS-aware client cache the ops proxy uses; a refusal is a chconn.HTTPError and a failure on the way wraps with %w, so the error classes keep working. An oversized response is typed and answers clickhouse.response_too_large. - Every read pins the rendering settings and sends wait_end_of_query, http_write_exception_in_output_format=0, max_execution_time (the tighter of the role's cap and query_timeout), cancellation on client close and readonly=2. A READONLY refusal of a read is clickhouse.misconfigured. - Write pipes run without readonly=2 and keep BYPASS, no-store and the never-retryable error answer. IsMutation stays, in sql_classify.go. - Reads share one connection cap per tenant (MaxConns, default 100). - The builder binds named {pN:String} parameters, an in list as one Array(String), refuses a null filter value and a value too large for the HTTP interface with a 400, and rewrites an RFC 3339 filter value only on a DateTime or DateTime64 column. - The cache key opens with a rendering marker, so builds that render differently never share a Redis entry. BREAKING CHANGE: /v1/query and pipe responses carry ClickHouse's rendering: keys in SELECT order, DateTime as `YYYY-MM-DD hh:mm:ss[.fff]` in the column's zone, decimals as numbers, NaN and Inf as null. A null filter value is a 400. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- go.mod | 2 +- internal/api/cache_key.go | 69 ++- internal/api/cache_key_test.go | 67 +-- internal/api/cache_tenant_test.go | 123 +++--- internal/api/ch_errors.go | 56 ++- internal/api/ch_errors_test.go | 173 ++++---- internal/api/ch_settings.go | 48 ++- internal/api/ch_settings_test.go | 41 +- internal/api/clickhouse_exec_test.go | 194 --------- internal/api/clickhouse_http.go | 319 ++++++++++++++ internal/api/clickhouse_http_test.go | 404 ++++++++++++++++++ internal/api/pipes.go | 85 ++-- internal/api/pipes_test.go | 100 +++-- .../{clickhouse_exec.go => sql_classify.go} | 128 +----- internal/api/sql_classify_test.go | 40 ++ internal/api/structured_query.go | 127 +++--- internal/api/structured_query_test.go | 238 ++++++++--- internal/api/tenant_clickhouse_test.go | 23 +- internal/api/tenant_helpers_test.go | 6 - internal/app/wire.go | 20 +- internal/pipes/pipes.go | 14 +- internal/pipes/pipes_test.go | 50 +-- internal/query/builder.go | 227 +++++++++- internal/query/builder_test.go | 189 +++++++- 24 files changed, 1861 insertions(+), 882 deletions(-) delete mode 100644 internal/api/clickhouse_exec_test.go create mode 100644 internal/api/clickhouse_http.go create mode 100644 internal/api/clickhouse_http_test.go rename internal/api/{clickhouse_exec.go => sql_classify.go} (68%) create mode 100644 internal/api/sql_classify_test.go diff --git a/go.mod b/go.mod index 9ff8605e..62d31dba 100644 --- a/go.mod +++ b/go.mod @@ -29,7 +29,6 @@ require ( github.com/fsnotify/fsnotify v1.10.1 github.com/go-chi/chi/v5 v5.3.2 github.com/golang-jwt/jwt/v5 v5.3.1 - github.com/google/uuid v1.6.0 github.com/ilyakaznacheev/cleanenv v1.5.0 github.com/klauspost/compress v1.19.2 github.com/moby/moby/api v1.55.0 @@ -151,6 +150,7 @@ require ( github.com/google/go-tpm v0.9.8 // indirect github.com/google/shlex v0.0.0-20191202100458-e7afc7fbc510 // indirect github.com/google/subcommands v1.2.0 // indirect + github.com/google/uuid v1.6.0 // indirect github.com/grpc-ecosystem/grpc-gateway/v2 v2.30.0 // indirect github.com/jedib0t/go-pretty/v6 v6.7.10 // indirect github.com/jmespath/go-jmespath v0.4.0 // indirect diff --git a/internal/api/cache_key.go b/internal/api/cache_key.go index 028bb778..0319e077 100644 --- a/internal/api/cache_key.go +++ b/internal/api/cache_key.go @@ -4,8 +4,7 @@ import ( "crypto/sha256" "encoding/binary" "encoding/hex" - "encoding/json" - "fmt" + "hash" "github.com/Wave-RF/WaveHouse/internal/tenant" ) @@ -25,44 +24,38 @@ import ( // answer one tenant with another's rows now that each tenant reads its own // ClickHouse (story 6). // -// Every section is framed with a 1-byte type marker (0x01 for sql, 0x00 for -// param) plus an 8-byte big-endian length, then the payload. Without -// length-prefixing the sql itself, a SQL string crafted to end with the -// exact bytes of a param frame (`\x00` + 8 length bytes + payload) would -// hash identically to a shorter SQL plus a real param — distinct -// `(sql, params)` tuples, same digest. Per-param framing also kept its -// own length prefix so embedded `\x00` inside a string param can't be -// confused for a frame boundary, and the JSON `{type, value}` payload -// shape additionally separates `"42"` (string) from `42` (int) so the -// cache can't serve a string-typed row to an int-typed lookup. -func queryCacheKey(id tenant.ID, sql string, params []any) string { +// The hash opens with chRendering, the response contract the cached bytes +// were rendered under, so a build that renders rows differently never serves +// another's entries from a shared cache — even for a pipe or a query without +// parameters, whose SQL is the same under both. +// +// params are the ClickHouse query-parameter values, positionally, exactly as +// they go on the wire: already rendered as text and encoded for ClickHouse's +// parameter reader. Two reads whose parameters reach ClickHouse as the same +// bytes are the same query, and two that do not are not. +// +// Every section is framed with a 1-byte type marker (0x02 for the rendering, +// 0x01 for sql, 0x00 for a param) plus an 8-byte big-endian length, then the +// payload. Without length-prefixing the sql itself, a SQL string crafted to +// end with the exact bytes of a param frame (`\x00` + 8 length bytes + +// payload) would hash identically to a shorter SQL plus a real param — +// distinct `(sql, params)` tuples, same digest. Per-param framing keeps its +// own length prefix so an embedded `\x00` inside a value can't be confused +// for a frame boundary. +func queryCacheKey(id tenant.ID, sql string, params []string) string { h := sha256.New() - var sqlLen [8]byte - binary.BigEndian.PutUint64(sqlLen[:], uint64(len(sql))) - _, _ = h.Write([]byte{1}) // sql frame marker - _, _ = h.Write(sqlLen[:]) - _, _ = h.Write([]byte(sql)) + writeFrame(h, 2, chRendering) + writeFrame(h, 1, sql) for _, p := range params { - payload, err := json.Marshal(struct { - Type string `json:"type"` - Value any `json:"value"` - }{ - Type: fmt.Sprintf("%T", p), - Value: p, - }) - if err != nil { - // Marshal can only fail on unserialisable types (channels, - // funcs, cyclic structures) that shouldn't reach this path — - // pipes/structured-query params are scalars from JSON. Fall - // back to a type+%v rendering so we still produce a key and - // don't take down the request path. - payload = fmt.Appendf(nil, "%T:%v", p, p) - } - var n [8]byte - binary.BigEndian.PutUint64(n[:], uint64(len(payload))) - _, _ = h.Write([]byte{0}) // param frame marker - _, _ = h.Write(n[:]) - _, _ = h.Write(payload) + writeFrame(h, 0, p) } return id.String() + ":query:" + hex.EncodeToString(h.Sum(nil)) } + +func writeFrame(h hash.Hash, marker byte, payload string) { + var n [8]byte + binary.BigEndian.PutUint64(n[:], uint64(len(payload))) + _, _ = h.Write([]byte{marker}) + _, _ = h.Write(n[:]) + _, _ = h.Write([]byte(payload)) +} diff --git a/internal/api/cache_key_test.go b/internal/api/cache_key_test.go index 44338316..aafe263d 100644 --- a/internal/api/cache_key_test.go +++ b/internal/api/cache_key_test.go @@ -1,6 +1,8 @@ package api import ( + "crypto/sha256" + "encoding/hex" "strings" "testing" @@ -18,9 +20,9 @@ func TestQueryCacheKey(t *testing.T) { tests := []struct { name string sqlA string - paramsA []any + paramsA []string sqlB string - paramsB []any + paramsB []string expectEqual bool }{ { @@ -44,49 +46,46 @@ func TestQueryCacheKey(t *testing.T) { sqlA: "SELECT 1", paramsA: nil, sqlB: "SELECT 1", - paramsB: []any{"a"}, + paramsB: []string{"a"}, expectEqual: false, }, { name: "embedded NUL byte does not collide with split params", sqlA: "SELECT 1", - paramsA: []any{"foo\x00bar"}, + paramsA: []string{"foo\x00bar"}, sqlB: "SELECT 1", - paramsB: []any{"foo", "bar"}, + paramsB: []string{"foo", "bar"}, expectEqual: false, }, { - name: "string and int with same textual value are distinct", - sqlA: "SELECT 1", - paramsA: []any{"42"}, - sqlB: "SELECT 1", - paramsB: []any{42}, - expectEqual: false, + // Every value reaches ClickHouse as a String parameter, so the + // same text is the same query whatever JSON type it arrived as. + name: "the same text is the same key", + sqlA: "SELECT {p0:String}", + paramsA: []string{"42"}, + sqlB: "SELECT {p0:String}", + paramsB: []string{"42"}, + expectEqual: true, }, { name: "nil and empty slice params produce the same key", sqlA: "SELECT 1", paramsA: nil, sqlB: "SELECT 1", - paramsB: []any{}, + paramsB: []string{}, expectEqual: true, }, { - // Constructed as an actual collision pair under the old - // "raw sql + framed params" format. The param frame for - // `"y"` is 0x00 + 8-byte BE length (0x1D = 29) + - // `{"type":"string","value":"y"}` (29 bytes), so the byte - // stream `("X", ["y"])` produces under the old framing is - // `"X" + 0x00 + 0x00…0x1D + {"type":"string","value":"y"}`. - // Setting sqlA to exactly those bytes and paramsA to nil - // reproduces that stream — under the old framing the two - // inputs hashed identically. The new 0x01-marker + 8-byte - // length prefix on sql forces them apart. + // Constructed as a collision pair under a "raw sql + framed + // params" format: the param frame for "y" is 0x00 + the 8-byte + // big-endian length 1 + "y", so sqlA with no params produces the + // byte stream ("X", ["y"]) would. The 0x01 marker and length + // prefix on the sql force them apart. name: "sql crafted to mimic a param-frame stream does not collide with shorter sql + real param", - sqlA: "X\x00\x00\x00\x00\x00\x00\x00\x00\x1d{\"type\":\"string\",\"value\":\"y\"}", + sqlA: "X\x00\x00\x00\x00\x00\x00\x00\x00\x01y", paramsA: nil, sqlB: "X", - paramsB: []any{"y"}, + paramsB: []string{"y"}, expectEqual: false, }, } @@ -110,8 +109,8 @@ func TestQueryCacheKey(t *testing.T) { // the flat directory's tenant 0 simply gains its "0". func TestQueryCacheKey_LeadsWithTheTenant(t *testing.T) { t.Parallel() - acme := queryCacheKey("acme", "SELECT 1", []any{"a"}) - globex := queryCacheKey("globex", "SELECT 1", []any{"a"}) + acme := queryCacheKey("acme", "SELECT 1", []string{"a"}) + globex := queryCacheKey("globex", "SELECT 1", []string{"a"}) assert.True(t, strings.HasPrefix(acme, "acme:query:"), acme) assert.True(t, strings.HasPrefix(globex, "globex:query:"), globex) assert.NotEqual(t, acme, globex) @@ -119,3 +118,19 @@ func TestQueryCacheKey_LeadsWithTheTenant(t *testing.T) { "the tenant is a prefix, not an input to the hash") assert.True(t, strings.HasPrefix(queryCacheKey(tenant.Default, "SELECT 1", nil), "0:query:")) } + +// The rendering contract opens the hash: a build that renders rows another +// way (chRendering) keys every entry apart from this one's, including a pipe +// or a query with no parameters, whose SQL is the same under both builds. +func TestQueryCacheKey_LeadsWithTheRendering(t *testing.T) { + t.Parallel() + h := sha256.New() + writeFrame(h, 2, chRendering) + writeFrame(h, 1, "SELECT 1") + assert.Equal(t, "0:query:"+hex.EncodeToString(h.Sum(nil)), queryCacheKey(tenant.Default, "SELECT 1", nil)) + + other := sha256.New() + writeFrame(other, 2, "JSON/0") + writeFrame(other, 1, "SELECT 1") + assert.NotEqual(t, "0:query:"+hex.EncodeToString(other.Sum(nil)), queryCacheKey(tenant.Default, "SELECT 1", nil)) +} diff --git a/internal/api/cache_tenant_test.go b/internal/api/cache_tenant_test.go index 3b335e1a..16ca37f7 100644 --- a/internal/api/cache_tenant_test.go +++ b/internal/api/cache_tenant_test.go @@ -11,11 +11,11 @@ import ( "testing/synctest" "time" - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" "github.com/Wave-RF/WaveHouse/internal/cache" + "github.com/Wave-RF/WaveHouse/internal/chconn" "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/pipes" "github.com/Wave-RF/WaveHouse/internal/policy" @@ -26,31 +26,14 @@ import ( "github.com/Wave-RF/WaveHouse/internal/testutil" ) -// countingConn answers every query with an empty result set and counts them, -// so a test can tell a cache hit (no query) from a miss (one query). -type countingConn struct { - driver.Conn - queries atomic.Int32 -} - -func (c *countingConn) Query(context.Context, string, ...any) (driver.Rows, error) { - c.queries.Add(1) - return &chainEmptyRows{}, nil -} - -// gatedConn holds every query open until release is closed and reports each -// one as it starts, so a test can hold requests in flight together and count -// the queries they became. -type gatedConn struct { - driver.Conn - entered chan struct{} // one send per query as it starts - release chan struct{} // closed to let every query finish -} - -func (c *gatedConn) Query(context.Context, string, ...any) (driver.Rows, error) { - c.entered <- struct{}{} - <-c.release - return &chainEmptyRows{}, nil +// gatedCH is a fakeCH that holds every query open until release is closed +// and reports each one on entered as it starts, so a test can hold requests +// in flight together and count the queries they became. +func gatedCH(entered, release chan struct{}) *fakeCH { + return &fakeCH{answer: func(http.ResponseWriter, *chSeen) { + entered <- struct{}{} + <-release + }} } // cachedRoutes are the two read paths that cache and coalesce, each with a @@ -61,16 +44,16 @@ var cachedRoutes = []struct{ name, path, body string }{ } // cachedRouter is the real router over tenants, both cached read paths wired -// to conn and c. Every request resolves to the viewer role, which may read +// to ch and c. Every request resolves to the viewer role, which may read // clicks.page and run top_pages. -func cachedRouter(t *testing.T, tenants *settings.Registry, conn driver.Conn, c cache.Cache) http.Handler { +func cachedRouter(t *testing.T, tenants *settings.Registry, ch *fakeCH, c cache.Cache) http.Handler { t.Helper() - return cachedRouterOver(t, tenants, fixedConn(conn), c) + return cachedRouterOver(t, tenants, ch.target, ch, c) } -// cachedRouterOver is cachedRouter with the tenant's connection chosen per -// request by connFor. -func cachedRouterOver(t *testing.T, tenants *settings.Registry, connFor func(*settings.Store) driver.Conn, c cache.Cache) http.Handler { +// cachedRouterOver is cachedRouter with the tenant's ClickHouse target chosen +// per request by targetFor. +func cachedRouterOver(t *testing.T, tenants *settings.Registry, targetFor func(*settings.Store) chconn.Target, ch *fakeCH, c cache.Cache) http.Handler { t.Helper() reg := testRegistry(t) viewer := staticPolicy(&policy.Policy{ @@ -78,11 +61,15 @@ func cachedRouterOver(t *testing.T, tenants *settings.Registry, connFor func(*se Tables: map[string]policy.TablePolicy{"clicks": {"viewer": {Select: &policy.SelectPermissions{AllowColumns: []string{"page"}}}}}, }) timeout := func(*settings.Store) time.Duration { return 5 * time.Second } + sq := NewStructuredQueryHandler(targetFor, c, fixedRegistry(reg), viewer, func(*settings.Store) int { return 60 }, timeout, nil) + sq.ch = ch.reader() + pipesHandler := NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), viewer, targetFor, c, timeout) + pipesHandler.ch = ch.reader() return NewRouter(Dependencies{ Tenants: tenants, Ingest: NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}), - StructuredQuery: NewStructuredQueryHandler(connFor, c, fixedRegistry(reg), viewer, func(*settings.Store) int { return 60 }, timeout, nil), - Pipes: NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), viewer, connFor, c, timeout), + StructuredQuery: sq, + Pipes: pipesHandler, Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, @@ -123,12 +110,12 @@ func TestNewRouter_CacheIsKeyedByTenant(t *testing.T) { l1, err := cache.NewLocal(1 << 20) require.NoError(t, err) t.Cleanup(func() { _ = l1.Close() }) - conn := &countingConn{} - router := cachedRouter(t, tenants, conn, l1) + ch := &fakeCH{} + router := cachedRouter(t, tenants, ch, l1) for _, route := range cachedRoutes { t.Run(route.name, func(t *testing.T) { - before := conn.queries.Load() + before := ch.reads.Load() // Ristretto admits asynchronously: settle after each request so the // next one reads what the last one stored. xcache := func(id tenant.ID) string { @@ -141,7 +128,7 @@ func TestNewRouter_CacheIsKeyedByTenant(t *testing.T) { assert.Equal(t, "HIT", xcache("acme")) assert.Equal(t, "MISS", xcache("globex"), "a tenant must never be served another tenant's cached result") assert.Equal(t, "HIT", xcache("globex")) - assert.Equal(t, before+2, conn.queries.Load(), "one query per tenant") + assert.Equal(t, before+2, ch.reads.Load(), "one query per tenant") }) } } @@ -154,12 +141,12 @@ func TestNewRouter_FlatDirectoryCacheStillHits(t *testing.T) { l1, err := cache.NewLocal(1 << 20) require.NoError(t, err) t.Cleanup(func() { _ = l1.Close() }) - conn := &countingConn{} - router := cachedRouter(t, testTenants(), conn, l1) + ch := &fakeCH{} + router := cachedRouter(t, testTenants(), ch, l1) for _, route := range cachedRoutes { t.Run(route.name, func(t *testing.T) { - before := conn.queries.Load() + before := ch.reads.Load() xcache := func(id tenant.ID) string { w := serveAs(t, router, route.path, route.body, id) require.Equal(t, http.StatusOK, w.Code, "body: %s", w.Body.String()) @@ -169,7 +156,7 @@ func TestNewRouter_FlatDirectoryCacheStillHits(t *testing.T) { assert.Equal(t, "MISS", xcache("")) assert.Equal(t, "HIT", xcache("")) assert.Equal(t, "HIT", xcache(tenant.Default)) - assert.Equal(t, before+1, conn.queries.Load()) + assert.Equal(t, before+1, ch.reads.Load()) }) } } @@ -194,8 +181,8 @@ func TestCachedRoutes_SingleflightIsPerTenant(t *testing.T) { for _, tt := range tests { t.Run(route.name+", "+tt.name, func(t *testing.T) { synctest.Test(t, func(t *testing.T) { - conn := &gatedConn{entered: make(chan struct{}, len(tt.tenants)), release: make(chan struct{})} - router := cachedRouter(t, tenants, conn, nil) + entered, release := make(chan struct{}, len(tt.tenants)), make(chan struct{}) + router := cachedRouter(t, tenants, gatedCH(entered, release), nil) var wg sync.WaitGroup for _, id := range tt.tenants { wg.Go(func() { @@ -204,8 +191,8 @@ func TestCachedRoutes_SingleflightIsPerTenant(t *testing.T) { }) } synctest.Wait() - assert.Equal(t, tt.wantQueries, len(conn.entered), "queries in flight once every request is blocked") - close(conn.release) + assert.Equal(t, tt.wantQueries, len(entered), "queries in flight once every request is blocked") + close(release) wg.Wait() }) }) @@ -213,21 +200,6 @@ func TestCachedRoutes_SingleflightIsPerTenant(t *testing.T) { } } -// bumpingConn runs bump inside the first query only, as an insert that lands -// while ClickHouse is still reading would. -type bumpingConn struct { - driver.Conn - bump func() - queries atomic.Int32 -} - -func (c *bumpingConn) Query(context.Context, string, ...any) (driver.Rows, error) { - if c.queries.Add(1) == 1 { - c.bump() - } - return &chainEmptyRows{}, nil -} - // #382: a result is filed under the versions read before its query ran, so // a bump landing mid-query orphans the fill — the next request misses and // reads the post-write rows — rather than serving pre-write rows until TTL. @@ -246,8 +218,15 @@ func TestCachedRoutes_BumpDuringQueryOrphansTheFill(t *testing.T) { l1, err := cache.NewLocal(1 << 20) require.NoError(t, err) t.Cleanup(func() { _ = l1.Close() }) - conn := &bumpingConn{bump: func() { require.NoError(t, bumps[route.name](t.Context(), l1)) }} - router := cachedRouter(t, testTenants(), conn, l1) + // The bump lands inside the first query only, as an insert that + // lands while ClickHouse is still reading would. + ch := &fakeCH{} + ch.answer = func(http.ResponseWriter, *chSeen) { + if ch.reads.Load() == 1 { + require.NoError(t, bumps[route.name](t.Context(), l1)) + } + } + router := cachedRouter(t, testTenants(), ch, l1) xcache := func() string { w := serveAs(t, router, route.path, route.body, "") require.Equal(t, http.StatusOK, w.Code, "body: %s", w.Body.String()) @@ -257,7 +236,7 @@ func TestCachedRoutes_BumpDuringQueryOrphansTheFill(t *testing.T) { assert.Equal(t, "MISS", xcache()) assert.Equal(t, "MISS", xcache(), "the fill of a query a bump overtook is orphaned") assert.Equal(t, "HIT", xcache()) - assert.Equal(t, int32(2), conn.queries.Load()) + assert.Equal(t, int32(2), ch.reads.Load()) }) } } @@ -274,19 +253,19 @@ func TestCachedRoutes_ReloadAsThePoolIsTakenOrphansTheFill(t *testing.T) { l1, err := cache.NewLocal(1 << 20) require.NoError(t, err) t.Cleanup(func() { _ = l1.Close() }) - conn := &countingConn{} + ch := &fakeCH{} var taken atomic.Int32 var noPool atomic.Bool - connFor := func(*settings.Store) driver.Conn { + targetFor := func(s *settings.Store) chconn.Target { if noPool.Load() { - return nil + return chconn.Target{} } if taken.Add(1) == 1 { require.NoError(t, l1.InvalidateTenant(t.Context(), tenant.Default)) } - return conn + return ch.target(s) } - router := cachedRouterOver(t, testTenants(), connFor, l1) + router := cachedRouterOver(t, testTenants(), targetFor, ch, l1) xcache := func() string { w := serveAs(t, router, route.path, route.body, "") require.Equal(t, http.StatusOK, w.Code, "body: %s", w.Body.String()) @@ -296,7 +275,7 @@ func TestCachedRoutes_ReloadAsThePoolIsTakenOrphansTheFill(t *testing.T) { assert.Equal(t, "MISS", xcache()) assert.Equal(t, "MISS", xcache(), "the fill of a query on the pool a reload replaced is orphaned") assert.Equal(t, "HIT", xcache()) - assert.Equal(t, int32(2), conn.queries.Load()) + assert.Equal(t, int32(2), ch.reads.Load()) noPool.Store(true) w := serveAs(t, router, route.path, route.body, "") @@ -322,9 +301,11 @@ func TestStructuredQuery_RawTableNameMeetsTheInsertsBump(t *testing.T) { l1, err := cache.NewLocal(1 << 20) require.NoError(t, err) t.Cleanup(func() { _ = l1.Close() }) - h := NewStructuredQueryHandler(fixedConn(&countingConn{}), l1, fixedRegistry(testutil.NewTestSchemaRegistry(t, schemas)), + ch := &fakeCH{} + h := NewStructuredQueryHandler(ch.target, l1, fixedRegistry(testutil.NewTestSchemaRegistry(t, schemas)), staticPolicy(&policy.Policy{DefaultRole: "viewer", Tables: grants}), func(*settings.Store) int { return 60 }, func(*settings.Store) time.Duration { return 5 * time.Second }, nil) + h.ch = ch.reader() for _, table := range tables { xcache := func() string { diff --git a/internal/api/ch_errors.go b/internal/api/ch_errors.go index adb61aaf..10ff1300 100644 --- a/internal/api/ch_errors.go +++ b/internal/api/ch_errors.go @@ -1,9 +1,10 @@ package api import ( - "context" + "errors" "log/slog" "net/http" + "strings" "time" "github.com/Wave-RF/WaveHouse/internal/chconn" @@ -31,7 +32,8 @@ const ( // query now. 503 with Retry-After — without it for a write pipe, which // may have run and is never retryable. codeCHUnavailable = "clickhouse.unavailable" - // codeCHResponseTooLarge: the raw-SQL proxy's response cap. 502. + // codeCHResponseTooLarge: the response outgrew the buffer every query + // path caps it at. 502, not retryable. codeCHResponseTooLarge = "clickhouse.response_too_large" // codeCHUnknown: a failure with no verdict. 5xx, retryable unless a // write pipe's. @@ -48,29 +50,31 @@ const ( chTooManyRows int32 = 158 chTimeoutExceeded int32 = 159 chTooSlow int32 = 160 + chReadonly int32 = 164 chMemoryLimit int32 = 241 chTooManyBytes int32 = 307 chTooManyRowsOrByte int32 = 396 chAccessDenied int32 = 497 ) -// capBackstop is how long past a role's time cap the client waits for -// ClickHouse's own TIMEOUT_EXCEEDED before giving up on the query. +// capBackstop is how long past the max_execution_time a read sends the +// client waits for ClickHouse's own TIMEOUT_EXCEEDED before giving up on the +// query: the server checks its budget between blocks, so it overshoots a +// little, and its answer says which limit stopped the query where a dropped +// connection says nothing. const capBackstop = 2 * time.Second -// cancelAfter is parent cancelled after d, with no deadline on it: the -// driver derives max_execution_time from a deadline, overriding the one -// the role's cap sends. -func cancelAfter(parent context.Context, d time.Duration) (context.Context, context.CancelFunc) { - ctx, cancel := context.WithCancel(parent) - t := time.AfterFunc(d, cancel) - return ctx, func() { t.Stop(); cancel() } -} - -// queryCaps says which of the role's own resource caps a query ran under, -// so exceeding one reads as the query's cost rather than an outage. +// queryCaps says what a query ran under: which of the role's own resource +// caps, so exceeding one reads as the query's cost rather than an outage, +// and whether it ran as a read under readonly=2. type queryCaps struct { time, memory bool + // readonly marks a read sent with readonly=2. ClickHouse refusing it + // as READONLY means the statement writes — a pipe IsMutation reads as a + // read — or the user's profile is readonly=1 and refuses the settings + // every read sends: either way the configuration, not the caller, and + // the same again on a retry. + readonly bool } // chFailure is how one failed ClickHouse call answers. @@ -82,10 +86,16 @@ type chFailure struct { // chFailureOf maps a failed ClickHouse call onto the response, by the class // chconn.Classify gives it. unknownStatus is the status of a failure with no -// verdict: 502 on the proxy, whose upstream answered with something it -// could not class, 500 on the native paths. +// verdict: 502 on the proxy, which forwards whatever its upstream answered, +// 500 on the structured query and pipes. func chFailureOf(err error, unknownStatus int, caps queryCaps) chFailure { + if _, ok := errors.AsType[*chResponseTooLargeError](err); ok { + return chFailure{http.StatusBadGateway, codeCHResponseTooLarge, false} + } code, hasCode := chconn.ExceptionCode(err) + if caps.readonly && hasCode && code == chReadonly { + return chFailure{http.StatusBadGateway, codeCHMisconfigured, false} + } switch { case hasCode && (code == chTooManyRows || code == chTooManyBytes || code == chTooManyRowsOrByte), caps.time && hasCode && (code == chTimeoutExceeded || code == chTooSlow), @@ -133,6 +143,18 @@ func writeCHWriteError(w http.ResponseWriter, r *http.Request, err error, messag writeCHFailure(w, r, err, message, f) } +// chErrorMessage is what a failed ClickHouse call tells the caller: +// ClickHouse's own text, verbatim, for a refusal it answered, as the proxy +// forwards it; the error itself for anything else. +func chErrorMessage(err error) string { + if he, ok := errors.AsType[*chconn.HTTPError](err); ok { + if msg := strings.TrimSpace(he.Body); msg != "" { + return msg + } + } + return err.Error() +} + func writeCHFailure(w http.ResponseWriter, r *http.Request, err error, message string, f chFailure) { switch f.code { case codeCHUnavailable: diff --git a/internal/api/ch_errors_test.go b/internal/api/ch_errors_test.go index ee2df2db..008f3e62 100644 --- a/internal/api/ch_errors_test.go +++ b/internal/api/ch_errors_test.go @@ -3,11 +3,10 @@ package api import ( "context" "encoding/json" - "errors" - "fmt" "net" "net/http" "net/http/httptest" + "strings" "testing" "time" @@ -25,7 +24,8 @@ import ( "github.com/Wave-RF/WaveHouse/internal/tenant" ) -// failingConn answers every query with err, as the driver would. +// failingConn answers every query with err, as the driver would — schema +// discovery's connection, which still speaks the native protocol. type failingConn struct { driver.Conn err error @@ -56,45 +56,74 @@ func refusedDial(t *testing.T) error { return err } +// chErrorCase is one way ClickHouse, or the way to it, fails a query: the +// answer the fake gives (or the transport failure it meets), and the +// response that must come of it. type chErrorCase struct { - name string - err error - caps policy.SelectPermissions + name string + // answer is what ClickHouse sends; transport, when set, is the failure + // the request meets instead. + answer func(http.ResponseWriter, *chSeen) + transport error + // maxResponse, when set, caps the response buffer. + maxResponse int64 + caps policy.SelectPermissions + // readOnly marks a case that only a read meets: ClickHouse refusing + // readonly=2, which a write is never sent under. + readOnly bool wantStatus int wantCode string wantRetryable bool } +// refused is ClickHouse refusing a statement with code at status. +func refused(status int, code int32, msg string) func(http.ResponseWriter, *chSeen) { + return answerException(status, code, chExceptionBody(code, msg)) +} + func chErrorCases(t *testing.T) []chErrorCase { t.Helper() return []chErrorCase{ - {name: "syntax error", err: chException(62, "DB::Exception", "Syntax error"), wantStatus: 400, wantCode: codeCHRejected}, - {name: "unknown identifier (#271)", err: chException(47, "DB::Exception", "Unknown expression identifier `received_timestamp`"), wantStatus: 400, wantCode: codeCHRejected}, - {name: "type mismatch", err: chException(53, "DB::Exception", "Type mismatch"), wantStatus: 400, wantCode: codeCHRejected}, - {name: "unknown table", err: chException(60, "DB::Exception", "Table default.gone does not exist"), wantStatus: 400, wantCode: codeCHRejected}, - {name: "rows read cap", err: chException(158, "DB::Exception", "Limit for rows exceeded"), wantStatus: 400, wantCode: codeCHLimitExceeded}, - {name: "time cap of the role", err: chException(159, "DB::Exception", "Timeout exceeded"), caps: policy.SelectPermissions{MaxExecutionTime: 1}, wantStatus: 400, wantCode: codeCHLimitExceeded}, - // A bare deadline is a pool wait or a dial timeout under a capped - // role, not the cap: ClickHouse reports the cap itself as 159. - {name: "pool wait under a time cap", err: fmt.Errorf("clickhouse query: %w", context.DeadlineExceeded), caps: policy.SelectPermissions{MaxExecutionTime: 1}, wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, - {name: "backstop cancel under a time cap", err: fmt.Errorf("clickhouse query: %w", context.Canceled), caps: policy.SelectPermissions{MaxExecutionTime: 1}, wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, + {name: "syntax error", answer: refused(400, 62, "Syntax error"), wantStatus: 400, wantCode: codeCHRejected}, + {name: "unknown identifier (#271)", answer: refused(404, 47, "Unknown expression identifier `received_timestamp`"), wantStatus: 400, wantCode: codeCHRejected}, + {name: "type mismatch", answer: refused(400, 53, "Cannot convert string '2026-01-15T10:30:00Z' to type DateTime"), wantStatus: 400, wantCode: codeCHRejected}, + {name: "unknown table", answer: refused(404, 60, "Table default.gone does not exist"), wantStatus: 400, wantCode: codeCHRejected}, + {name: "rows read cap", answer: refused(500, 158, "Limit for rows exceeded"), wantStatus: 400, wantCode: codeCHLimitExceeded}, + {name: "time cap of the role", answer: refused(408, 159, "Timeout exceeded"), caps: policy.SelectPermissions{MaxExecutionTime: 1}, wantStatus: 400, wantCode: codeCHLimitExceeded}, + // A dropped connection under a capped role is not the cap: ClickHouse + // reports the cap itself as 159. + {name: "deadline under a time cap", transport: context.DeadlineExceeded, caps: policy.SelectPermissions{MaxExecutionTime: 1}, wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, + {name: "cancel under a time cap", transport: context.Canceled, caps: policy.SelectPermissions{MaxExecutionTime: 1}, wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, // A cap longer than the handler's 5s query_timeout is not what // stopped the query: the timeout reads as it does with no cap. - {name: "query_timeout under a longer time cap", err: chException(159, "DB::Exception", "Timeout exceeded"), caps: policy.SelectPermissions{MaxExecutionTime: 10000}, wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, - {name: "memory cap of the role", err: chException(241, "DB::Exception", "Memory limit (for query) exceeded"), caps: policy.SelectPermissions{MaxMemoryUsage: 1}, wantStatus: 400, wantCode: codeCHLimitExceeded}, - {name: "server timeout, no role cap", err: chException(159, "DB::Exception", "Timeout exceeded"), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, - {name: "server memory, no role cap", err: chException(241, "DB::Exception", "Memory limit (total) exceeded"), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, - {name: "missing grant", err: chException(497, "DB::Exception", "default: Not enough privileges"), wantStatus: 403, wantCode: codeCHAccessDenied}, - {name: "wrong password", err: chException(516, "DB::Exception", "Authentication failed"), wantStatus: 502, wantCode: codeCHMisconfigured}, - {name: "overloaded", err: chException(202, "DB::Exception", "Too many simultaneous queries"), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, - {name: "connection refused", err: fmt.Errorf("clickhouse query: %w", refusedDial(t)), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, - {name: "pool exhausted", err: fmt.Errorf("clickhouse query: %w", clickhouse.ErrAcquireConnTimeout), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, - {name: "codeless 404 on the way", err: fmt.Errorf("clickhouse query: %w", &clickhouse.HTTPError{StatusCode: 404, Err: errors.New("There is no handle /nope")}), wantStatus: 502, wantCode: codeCHMisconfigured}, - {name: "codeless 500 on the way", err: fmt.Errorf("clickhouse query: %w", &clickhouse.HTTPError{StatusCode: 500, Err: errors.New("upstream exploded")}), wantStatus: 500, wantCode: codeCHUnknown, wantRetryable: true}, - {name: "no verdict", err: errors.New("scan clickhouse row: something odd"), wantStatus: 500, wantCode: codeCHUnknown, wantRetryable: true}, + {name: "query_timeout under a longer time cap", answer: refused(408, 159, "Timeout exceeded"), caps: policy.SelectPermissions{MaxExecutionTime: 10000}, wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, + {name: "memory cap of the role", answer: refused(500, 241, "Memory limit (for query) exceeded"), caps: policy.SelectPermissions{MaxMemoryUsage: 1}, wantStatus: 400, wantCode: codeCHLimitExceeded}, + {name: "server timeout, no role cap", answer: refused(408, 159, "Timeout exceeded"), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, + {name: "server memory, no role cap", answer: refused(500, 241, "Memory limit (total) exceeded"), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, + {name: "missing grant", answer: refused(403, 497, "default: Not enough privileges"), wantStatus: 403, wantCode: codeCHAccessDenied}, + {name: "wrong password", answer: refused(403, 516, "Authentication failed"), wantStatus: 502, wantCode: codeCHMisconfigured}, + {name: "overloaded", answer: refused(500, 202, "Too many simultaneous queries"), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, + // A statement that writes, refused by the readonly=2 every read + // carries, or a readonly=1 profile refusing the settings every read + // sends: configuration, the same on a retry. + {name: "read-only refusal of a read", answer: refused(500, 164, "default: Cannot execute query in readonly mode"), readOnly: true, wantStatus: 502, wantCode: codeCHMisconfigured}, + {name: "connection refused", transport: refusedDial(t), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, + {name: "codeless 404 on the way", answer: answerException(404, 0, "There is no handle /nope"), wantStatus: 502, wantCode: codeCHMisconfigured}, + {name: "codeless 500 on the way", answer: answerException(500, 0, "upstream exploded"), wantStatus: 500, wantCode: codeCHUnknown, wantRetryable: true}, + {name: "exception after the rows", answer: answerRows("{\"page\":\"/\"}\n" + chExceptionBody(241, "Memory limit (total) exceeded")), wantStatus: 503, wantCode: codeCHUnavailable, wantRetryable: true}, + {name: "no verdict", answer: answerRows("{\"page\":\"/\"}\nsomething odd\n"), wantStatus: 500, wantCode: codeCHUnknown, wantRetryable: true}, + {name: "response too large", answer: answerRows(strings.Repeat("{\"page\":\"/\"}\n", 10)), maxResponse: 16, wantStatus: 502, wantCode: codeCHResponseTooLarge}, } } +// fake is the fakeCH tc answers through, and its reader. +func (tc chErrorCase) fake() (*fakeCH, *chReader) { + f := &fakeCH{answer: tc.answer, err: tc.transport} + r := f.reader() + r.maxResponseBytes = tc.maxResponse + return f, r +} + func assertCHError(t *testing.T, w *httptest.ResponseRecorder, tc chErrorCase) { t.Helper() require.Equal(t, tc.wantStatus, w.Code, w.Body.String()) @@ -120,7 +149,9 @@ func TestStructuredQuery_ClickHouseErrors(t *testing.T) { t.Parallel() perms := tc.caps perms.AllowColumns = []string{"*"} - h := newCapturingHandler(t, &failingConn{err: tc.err}, policyWithViewer(perms)) + f, r := tc.fake() + h := newCapturingHandler(t, f, policyWithViewer(perms)) + h.ch = r w := httptest.NewRecorder() h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{Columns: []string{"page"}}))) assertCHError(t, w, tc) @@ -138,12 +169,14 @@ func TestPipes_ClickHouseErrors(t *testing.T) { } t.Run(tc.name, func(t *testing.T) { t.Parallel() + f, r := tc.fake() store := staticPipes(&pipes.NamedQuery{Name: "p", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}) - h := NewPipesHandler(store, staticPolicy(&policy.Policy{}), fixedConn(&failingConn{err: tc.err}), nil, func(*settings.Store) time.Duration { return time.Second }) - r := pipesRequest(t, http.MethodGet, "/v1/pipes/p", "p", nil) - r = r.WithContext(auth.WithRole(r.Context(), "viewer")) + h := NewPipesHandler(store, staticPolicy(&policy.Policy{}), f.target, nil, func(*settings.Store) time.Duration { return time.Second }) + h.ch = r + req := pipesRequest(t, http.MethodGet, "/v1/pipes/p", "p", nil) + req = req.WithContext(auth.WithRole(req.Context(), "viewer")) w := httptest.NewRecorder() - h.Execute(w, withTenant(r)) + h.Execute(w, withTenant(req)) assertCHError(t, w, tc) }) } @@ -155,13 +188,14 @@ func TestPipes_ClickHouseErrors(t *testing.T) { func TestPipes_WriteClickHouseErrors(t *testing.T) { t.Parallel() for _, tc := range chErrorCases(t) { - if tc.caps.MaxExecutionTime > 0 || tc.caps.MaxMemoryUsage > 0 { + if tc.caps.MaxExecutionTime > 0 || tc.caps.MaxMemoryUsage > 0 || tc.readOnly { continue } t.Run(tc.name, func(t *testing.T) { t.Parallel() - conn := &writeConn{err: tc.err} - h := writerPipesHandler(t, conn, nil, &pipes.NamedQuery{Name: "log", SQL: "INSERT INTO audit_log VALUES ({{msg}}, now())"}) + f, r := tc.fake() + h := writerPipesHandler(t, f, nil, &pipes.NamedQuery{Name: "log", SQL: "INSERT INTO audit_log VALUES ({{msg}}, now())"}) + h.ch = r w := pipeCallAs(t, h, "log") require.Equal(t, tc.wantStatus, w.Code, w.Body.String()) var got errorBody @@ -170,11 +204,24 @@ func TestPipes_WriteClickHouseErrors(t *testing.T) { require.NotNil(t, got.Retryable) assert.False(t, *got.Retryable) assert.Empty(t, w.Header().Get("Retry-After")) - assert.Equal(t, int32(1), conn.execs.Load()) + assert.Equal(t, int32(1), f.writes.Load()) }) } } +// TestCHErrors_MessageIsClickHousesOwn: the caller reads ClickHouse's own +// refusal text, verbatim, as the proxy forwards it — not a wrapper around it. +func TestCHErrors_MessageIsClickHousesOwn(t *testing.T) { + t.Parallel() + f := &fakeCH{answer: refused(400, 62, "Syntax error")} + h := newCapturingHandler(t, f, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{Columns: []string{"page"}}))) + var got errorBody + require.NoError(t, json.Unmarshal(w.Body.Bytes(), &got)) + assert.Equal(t, strings.TrimSpace(chExceptionBody(62, "Syntax error")), got.Error) +} + type errRow struct{ err error } func (r errRow) Err() error { return r.err } @@ -200,55 +247,3 @@ func TestSchemaRefresh_ClickHouseDown(t *testing.T) { w := refresh(chException(62, "DB::Exception", "Syntax error")) require.Equal(t, http.StatusInternalServerError, w.Code, w.Body.String()) } - -// deadlineConn records whether the query context carried a deadline. -type deadlineConn struct { - driver.Conn - hasDeadline bool -} - -func (c *deadlineConn) Query(ctx context.Context, _ string, _ ...any) (driver.Rows, error) { - _, c.hasDeadline = ctx.Deadline() - return &chainEmptyRows{}, nil -} - -// TestStructuredQuery_TimeCapLeavesNoDeadline: under a role's time cap the -// query context has no deadline, so clickhouse-go keeps the cap's -// max_execution_time and an overrun comes back as TIMEOUT_EXCEEDED rather -// than a bare DeadlineExceeded. Without a cap, the query timeout is a -// deadline as before. -func TestStructuredQuery_TimeCapLeavesNoDeadline(t *testing.T) { - t.Parallel() - for _, tc := range []struct { - name string - perms policy.SelectPermissions - wantDeadline bool - }{ - {"time cap", policy.SelectPermissions{AllowColumns: []string{"*"}, MaxExecutionTime: 5000}, false}, - {"no time cap", policy.SelectPermissions{AllowColumns: []string{"*"}}, true}, - } { - t.Run(tc.name, func(t *testing.T) { - t.Parallel() - conn := &deadlineConn{} - h := newCapturingHandler(t, conn, policyWithViewer(tc.perms)) - w := httptest.NewRecorder() - h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{Columns: []string{"page"}}))) - require.Equal(t, http.StatusOK, w.Code, w.Body.String()) - assert.Equal(t, tc.wantDeadline, conn.hasDeadline) - }) - } -} - -func TestCancelAfter(t *testing.T) { - t.Parallel() - ctx, cancel := cancelAfter(t.Context(), 10*time.Millisecond) - defer cancel() - _, has := ctx.Deadline() - assert.False(t, has) - select { - case <-ctx.Done(): - case <-time.After(2 * time.Second): - t.Fatal("cancelAfter never cancelled") - } - assert.ErrorIs(t, ctx.Err(), context.Canceled) -} diff --git a/internal/api/ch_settings.go b/internal/api/ch_settings.go index b6b09aa0..e26200b6 100644 --- a/internal/api/ch_settings.go +++ b/internal/api/ch_settings.go @@ -1,23 +1,22 @@ package api import ( + "strconv" "time" - - "github.com/ClickHouse/clickhouse-go/v2" ) -// chQueryLimits is the per-request resource budget a single read runs under, -// taken from the role's resolved policy caps. A zero field means "no limit" for -// that dimension and is omitted from the settings. Server-wide backstops are -// ClickHouse's job (settings profiles / quotas), not WaveHouse's — so an admin, -// whose policy resolves to no caps, sends no settings here and is bounded only -// by ClickHouse's own config. +// chQueryLimits is the per-request budget a single read runs under: the +// role's resolved policy caps, and the time bound every read carries. A zero +// field means "no limit" for that dimension and is omitted from the settings. +// Server-wide backstops are ClickHouse's job (settings profiles / quotas), not +// WaveHouse's — so an admin, whose policy resolves to no caps, is bounded only +// by its query timeout and ClickHouse's own config. type chQueryLimits struct { // ExecutionTime is the wall-clock budget, emitted as max_execution_time in // fractional seconds, so ClickHouse itself stops the query and says so - // (TIMEOUT_EXCEEDED). The query context carries no deadline when this is - // set (cancelAfter): clickhouse-go would otherwise overwrite the setting - // with deadline+5s for any deadline over 1s. + // (TIMEOUT_EXCEEDED). Cancelling the HTTP request cannot interrupt a + // server-side phase already running, so the bound reaches ClickHouse as a + // setting rather than only as the request's deadline. ExecutionTime time.Duration // MaxResultRows caps rows RETURNED (max_result_rows + result_overflow_mode= // throw) — defense-in-depth behind the SQL LIMIT the structured builder @@ -31,29 +30,32 @@ type chQueryLimits struct { MaxMemoryBytes int64 } -// chReadSettings builds the per-query ClickHouse Settings that enforce a read's -// resource budget SERVER-SIDE, so it can't outrun the budget during a -// server-side scan / merge / aggregation phase (#316). Without these, the only -// budget reaching ClickHouse is whatever clickhouse-go derives from the context -// deadline — which never bounds memory or rows scanned. Returns nil when no cap -// applies, so the caller can skip wrapping the context. -func chReadSettings(l chQueryLimits) clickhouse.Settings { - settings := clickhouse.Settings{} +// chReadSettings builds the per-query ClickHouse settings that enforce a +// read's budget SERVER-SIDE, so it can't outrun the budget during a +// server-side scan / merge / aggregation phase (#316); the request's deadline +// alone never bounds memory or rows scanned. Returns nil when no limit +// applies. +// +// The values are text because they ride on the HTTP interface's query +// string, in the numeric spellings ClickHouse expects: max_execution_time in +// fractional seconds, the rest as plain integers. +func chReadSettings(l chQueryLimits) map[string]string { + settings := map[string]string{} if l.ExecutionTime > 0 { // Fractional seconds — ClickHouse accepts them, preserving a sub-second // cap that a whole-second representation would round away. - settings["max_execution_time"] = l.ExecutionTime.Seconds() + settings["max_execution_time"] = strconv.FormatFloat(l.ExecutionTime.Seconds(), 'f', -1, 64) } if l.MaxResultRows > 0 { - settings["max_result_rows"] = l.MaxResultRows + settings["max_result_rows"] = strconv.Itoa(l.MaxResultRows) settings["result_overflow_mode"] = "throw" } if l.MaxRowsToRead > 0 { - settings["max_rows_to_read"] = l.MaxRowsToRead + settings["max_rows_to_read"] = strconv.FormatInt(l.MaxRowsToRead, 10) settings["read_overflow_mode"] = "throw" } if l.MaxMemoryBytes > 0 { - settings["max_memory_usage"] = l.MaxMemoryBytes + settings["max_memory_usage"] = strconv.FormatInt(l.MaxMemoryBytes, 10) } if len(settings) == 0 { return nil diff --git a/internal/api/ch_settings_test.go b/internal/api/ch_settings_test.go index 930b5eec..51a37ac4 100644 --- a/internal/api/ch_settings_test.go +++ b/internal/api/ch_settings_test.go @@ -16,8 +16,8 @@ func TestChReadSettings(t *testing.T) { name string limits chQueryLimits // want is the exact settings map expected; nil means chReadSettings - // must return nil (no caps → no context wrapping). - want map[string]any + // must return nil (no limit → no settings). + want map[string]string }{ { name: "no caps set", @@ -27,35 +27,36 @@ func TestChReadSettings(t *testing.T) { { name: "sub-second execution time is a fractional max_execution_time", limits: chQueryLimits{ExecutionTime: 500 * time.Millisecond}, - // The driver only auto-derives max_execution_time for deadlines > 1s, - // so a 500ms cap MUST be emitted explicitly or it reaches CH unbounded. - want: map[string]any{"max_execution_time": 0.5}, + // A request deadline is not a server-side bound, so a sub-second cap + // MUST be emitted explicitly — and as fractional seconds, which a + // whole-second spelling would round away to "no cap at all". + want: map[string]string{"max_execution_time": "0.5"}, }, { name: "multi-second execution time", limits: chQueryLimits{ExecutionTime: 3 * time.Second}, - want: map[string]any{"max_execution_time": 3.0}, + want: map[string]string{"max_execution_time": "3"}, }, { name: "max_result_rows caps result rows with throw mode", limits: chQueryLimits{MaxResultRows: 1000}, - want: map[string]any{ - "max_result_rows": 1000, + want: map[string]string{ + "max_result_rows": "1000", "result_overflow_mode": "throw", }, }, { name: "max_rows_to_read caps rows scanned with throw mode", limits: chQueryLimits{MaxRowsToRead: 1_000_000}, - want: map[string]any{ - "max_rows_to_read": int64(1_000_000), + want: map[string]string{ + "max_rows_to_read": "1000000", "read_overflow_mode": "throw", }, }, { name: "max_memory_usage caps peak query memory", limits: chQueryLimits{MaxMemoryBytes: 4 << 30}, // 4 GiB > int32 - want: map[string]any{"max_memory_usage": int64(4 << 30)}, + want: map[string]string{"max_memory_usage": "4294967296"}, }, { name: "all caps together", @@ -65,20 +66,20 @@ func TestChReadSettings(t *testing.T) { MaxRowsToRead: 2_000_000, MaxMemoryBytes: 8 << 30, }, - want: map[string]any{ - "max_execution_time": 2.0, - "max_result_rows": 500, + want: map[string]string{ + "max_execution_time": "2", + "max_result_rows": "500", "result_overflow_mode": "throw", - "max_rows_to_read": int64(2_000_000), + "max_rows_to_read": "2000000", "read_overflow_mode": "throw", - "max_memory_usage": int64(8 << 30), + "max_memory_usage": "8589934592", }, }, { name: "zero caps are omitted even when others are set", limits: chQueryLimits{MaxRowsToRead: 42}, - want: map[string]any{ - "max_rows_to_read": int64(42), + want: map[string]string{ + "max_rows_to_read": "42", "read_overflow_mode": "throw", }, }, @@ -99,12 +100,12 @@ func TestChReadSettings(t *testing.T) { t.Fatalf("expected settings %#v, got nil", tt.want) } if len(got) != len(tt.want) { - t.Fatalf("settings key count mismatch: got %#v, want %#v", map[string]any(got), tt.want) + t.Fatalf("settings key count mismatch: got %#v, want %#v", got, tt.want) } for k, wantV := range tt.want { gotV, ok := got[k] if !ok { - t.Errorf("missing setting %q (got %#v)", k, map[string]any(got)) + t.Errorf("missing setting %q (got %#v)", k, got) continue } if gotV != wantV { diff --git a/internal/api/clickhouse_exec_test.go b/internal/api/clickhouse_exec_test.go deleted file mode 100644 index 1ead574f..00000000 --- a/internal/api/clickhouse_exec_test.go +++ /dev/null @@ -1,194 +0,0 @@ -package api - -import ( - "context" - "reflect" - "testing" - "time" - - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" - "github.com/Wave-RF/WaveHouse/internal/testutil/mutationtest" - "github.com/google/uuid" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -// stubConn records how many times Exec or Query was called and returns -// canned results. The embedded nil driver.Conn keeps every method we don't -// override undefined-method-call-panic'd, which is what we want — the test -// fails loudly if executeCHQuery starts touching new surface area. -type stubConn struct { - driver.Conn - execCount int - queryCount int - execErr error - // queryRows, when non-nil, is returned by Query in place of the default - // empty rows. Tests that exercise row-scan / transformRow paths use this. - queryRows driver.Rows -} - -func (c *stubConn) Exec(_ context.Context, _ string, _ ...any) error { - c.execCount++ - return c.execErr -} - -func (c *stubConn) Query(_ context.Context, _ string, _ ...any) (driver.Rows, error) { - c.queryCount++ - if c.queryRows != nil { - return c.queryRows, nil - } - return &chainEmptyRows{}, nil -} - -// TestIsMutation runs the shared cases; the integration suite checks the -// same cases against ClickHouse's parser. -func TestIsMutation(t *testing.T) { - t.Parallel() - for _, tc := range mutationtest.Cases { - t.Run(tc.Name, func(t *testing.T) { - t.Parallel() - assert.Equal(t, tc.Mutation, IsMutation(tc.SQL)) - }) - } -} - -// TestIsMutation_ClickHouseWhitespace pins every character ClickHouse 26.6's -// lexer accepts as whitespace, each checked against a live server: ahead of a -// write it must not hide the verb, and ahead of a read it must not make one. -func TestIsMutation_ClickHouseWhitespace(t *testing.T) { - t.Parallel() - spaces := []rune{' ', '\t', '\n', '\v', '\f', '\r', 0x85, 0xA0, 0x180E, 0x2028, 0x2029, 0x202F, 0x205F, 0x2060, 0x3000, 0xFEFF} - for r := rune(0x2000); r <= 0x200D; r++ { - spaces = append(spaces, r) - } - for _, r := range spaces { - ws := string(r) - assert.True(t, IsMutation(ws+"INSERT INTO t VALUES (1)"), "U+%04X before INSERT", r) - assert.False(t, IsMutation(ws+"SELECT 1"), "U+%04X before SELECT", r) - assert.True(t, IsMutation("WITH x AS (SELECT 1)"+ws+"INSERT INTO t SELECT * FROM x"), "U+%04X before a WITH's INSERT", r) - assert.True(t, IsMutation("WITH 1 AS x INSERT"+ws+"INTO t SELECT x"), "U+%04X between a WITH's INSERT and INTO", r) - } - // Not whitespace to ClickHouse (it rejects the statement), so not skipped. - assert.False(t, IsMutation("\u1680INSERT INTO t VALUES (1)")) -} - -func TestExecuteCHQuery_MutationRoutesToExec(t *testing.T) { - t.Parallel() - // Mutations route through driver.Exec because clickhouse-go's - // driver.Query() errors on statements that return no result set. - // executeCHQuery marshals the no-rows case to `[]` so the response shape - // stays "always an array" regardless of whether the SQL was a read or - // a mutation. Used by structured_query and pipes handlers; the raw-SQL - // handler bypasses this entirely (HTTP proxy). - for _, sql := range []string{ - "TRUNCATE TABLE clicks", - "DROP TABLE clicks", - "DELETE FROM clicks WHERE id = 1", - "ALTER TABLE clicks ADD COLUMN c String", - "INSERT INTO clicks VALUES (1)", - " -- audit log\n UPDATE clicks SET v = 1 WHERE id = 2", - "EXECUTE AS writer INSERT INTO clicks VALUES (1)", - } { - t.Run(sql, func(t *testing.T) { - t.Parallel() - conn := &stubConn{} - rows, err := executeCHQuery(context.Background(), conn, sql, nil) - require.NoError(t, err) - assert.Equal(t, 1, conn.execCount, "Exec must be used for mutations") - assert.Zero(t, conn.queryCount, "Query must not be used for mutations") - assert.Equal(t, []map[string]any{}, rows, "mutation result must marshal to [] not null") - }) - } -} - -func TestExecuteCHQuery_SelectRoutesToQuery(t *testing.T) { - t.Parallel() - for _, sql := range []string{"SELECT 1", "EXECUTE AS reader SELECT 1"} { - t.Run(sql, func(t *testing.T) { - t.Parallel() - conn := &stubConn{} - rows, err := executeCHQuery(context.Background(), conn, sql, nil) - require.NoError(t, err) - assert.Zero(t, conn.execCount, "Exec must not be used for SELECT") - assert.Equal(t, 1, conn.queryCount, "Query must be used for SELECT") - assert.Equal(t, []map[string]any{}, rows, "zero-row SELECT must marshal to [] not null") - }) - } -} - -// TestExecuteCHQuery_TransformsClickHouseTypes pins transformRow's contract -// at the unit level: UUIDs become canonical strings, time.Time — including the -// *time.Time a Nullable(DateTime…) column scans as — becomes RFC3339Nano UTC -// (NULL stays JSON null), and other scalars pass through unchanged. The -// integration suite exercises the same path against a real ClickHouse, but -// this unit test catches regressions in the type-conversion branches without -// standing up testcontainers. -func TestExecuteCHQuery_TransformsClickHouseTypes(t *testing.T) { - t.Parallel() - id := uuid.MustParse("11111111-2222-3333-4444-555555555555") - ts := time.Date(2026, 5, 19, 12, 30, 45, 123456789, time.FixedZone("EST", -5*3600)) - conn := &stubConn{queryRows: &chainOneRow{ - columns: []chainColumnType{ - {name: "id", scanType: reflect.TypeFor[uuid.UUID]()}, - {name: "received_at", scanType: reflect.TypeFor[time.Time]()}, - {name: "updated_at", scanType: reflect.TypeFor[*time.Time]()}, - {name: "deleted_at", scanType: reflect.TypeFor[*time.Time]()}, - {name: "n", scanType: reflect.TypeFor[int64]()}, - }, - values: []any{id, ts, &ts, (*time.Time)(nil), int64(42)}, - }} - - rows, err := executeCHQuery(context.Background(), conn, "SELECT id, received_at, updated_at, deleted_at, n FROM t", nil) - require.NoError(t, err) - require.Len(t, rows, 1) - assert.Equal(t, id.String(), rows[0]["id"], "UUID must be stringified") - assert.Equal(t, ts.UTC().Format(time.RFC3339Nano), rows[0]["received_at"], "time must be RFC3339Nano in UTC") - assert.Equal(t, ts.UTC().Format(time.RFC3339Nano), rows[0]["updated_at"], "nullable time must be RFC3339Nano in UTC") - assert.Nil(t, rows[0]["deleted_at"], "NULL nullable time must stay nil") - assert.Equal(t, int64(42), rows[0]["n"], "scalar must pass through unchanged") -} - -// chainOneRow implements driver.Rows for a single canned row. Scan -// reflect-writes values[i] into the i-th destination pointer that -// executeCHQuery allocates from ColumnTypes()[i].ScanType(). -type chainOneRow struct { - driver.Rows - columns []chainColumnType - values []any - yielded bool -} - -func (r *chainOneRow) Next() bool { - if r.yielded { - return false - } - r.yielded = true - return true -} - -func (r *chainOneRow) Scan(dest ...any) error { - for i, d := range dest { - reflect.ValueOf(d).Elem().Set(reflect.ValueOf(r.values[i])) - } - return nil -} - -func (*chainOneRow) Close() error { return nil } -func (*chainOneRow) Err() error { return nil } - -func (r *chainOneRow) ColumnTypes() []driver.ColumnType { - out := make([]driver.ColumnType, len(r.columns)) - for i := range r.columns { - out[i] = &r.columns[i] - } - return out -} - -type chainColumnType struct { - driver.ColumnType - name string - scanType reflect.Type -} - -func (c *chainColumnType) Name() string { return c.name } -func (c *chainColumnType) ScanType() reflect.Type { return c.scanType } diff --git a/internal/api/clickhouse_http.go b/internal/api/clickhouse_http.go new file mode 100644 index 00000000..1a0b82c8 --- /dev/null +++ b/internal/api/clickhouse_http.go @@ -0,0 +1,319 @@ +package api + +import ( + "bytes" + "context" + "crypto/tls" + "fmt" + "io" + "net/http" + "net/url" + "strconv" + "strings" + "sync" + "time" + + "github.com/Wave-RF/WaveHouse/internal/chconn" + "github.com/Wave-RF/WaveHouse/internal/settings" +) + +// chRendering names the response contract chReadSettingsFixed pins. It leads +// every cached read's key (queryCacheKey), so two builds that render rows +// differently never serve each other's entries from a shared cache during a +// rolling deploy. Change it with any change to the rendering settings below. +const chRendering = "JSONEachRow/1" + +// chReadSettingsFixed go on every request the cached read paths send, ahead +// of the role's caps. +// +// Rendering: ClickHouse renders the rows, so a Decimal's spelling, a +// DateTime64's scale, an Enum's name and an IPv6's compression are the +// server's own. Every knob that changes the bytes is pinned rather than +// inherited, because a tenant's server or profile may set any of them: +// 64-bit integers and decimals as bare numbers, NaN and Inf as null, +// DateTime as `YYYY-MM-DD hh:mm:ss[.fff]` (the spelling the SSE wire carries, +// #372), and a named tuple as an object. +// +// Failure: wait_end_of_query buffers the result server-side until the query +// has finished, and http_write_exception_in_output_format=0 keeps an +// exception out of the JSON, so a query that fails part way answers with a +// non-200 and X-ClickHouse-Exception-Code instead of a 200 carrying rows and +// then an error. Measured on 24.8.14.39 and 26.6.3.62: without them a failure +// after the first block arrived as rows followed by exception text (with a +// 200 on 24.8). +// +// Cancellation: the request's own deadline drops the connection, and +// cancel_http_readonly_queries_on_client_close stops the query on the server +// rather than letting it run on after nobody waits for it. +// +// Every one of these is known to ClickHouse 24.8 and later (measured on +// 24.8.14.39 and 26.6.3.62). +var chReadSettingsFixed = map[string]string{ + "default_format": "JSONEachRow", + "wait_end_of_query": "1", + "http_write_exception_in_output_format": "0", + "cancel_http_readonly_queries_on_client_close": "1", + "output_format_json_quote_64bit_integers": "0", + "output_format_json_quote_decimals": "0", + "output_format_json_quote_denormals": "0", + "date_time_output_format": "simple", + "output_format_json_named_tuples_as_objects": "1", +} + +// ClickHouse's own limits on the HTTP interface's query string, at their +// defaults: http_max_field_value_size per field and http_max_uri_size for the +// whole request line. Both count the percent-encoded text. Measured on +// 24.8.14.39 and 26.6.3.62: a 131072-byte value is read and a 132096-byte one +// is refused with code 1000 ("Field value too long"), and a 1 MiB value with +// a codeless 400 — answers chFailureOf would read as an outage and a +// misconfiguration rather than as a request that is too large. +const ( + chMaxFieldBytes = 128 << 10 + chMaxURIBytes = 1 << 20 + // chURIHeadroom is what the path, the settings and the database leave + // of chMaxURIBytes for the bound values. + chURIHeadroom = 16 << 10 +) + +// defaultReadConns caps a tenant's concurrent reads when it names no +// clickhouse.max_open_conns, as the ingest worker's transport does. +const defaultReadConns = 100 + +// chRequest is one statement for ClickHouse's HTTP interface. +type chRequest struct { + sql string + // params supply param_p0 … param_pN-1, positionally, for the {pN:…} + // placeholders query.BuildResult.NamedParams emitted, already encoded + // for ClickHouse's parameter reader. + params []string + // settings are per-query ClickHouse settings (chReadSettings). + settings map[string]string + // write runs the statement without readonly=2: a write pipe. Everything + // else is a read, and ClickHouse refuses it if it writes after all. + write bool +} + +// chReader runs the cached read paths — the structured query and named pipes +// — and write pipes against ClickHouse's HTTP interface, and hands back +// ClickHouse's own JSON rendering of the rows as a JSON array. +type chReader struct { + // build makes the client for a TLS config and a cap on connections: + // readerHTTPClient in production, a fake transport in tests. + build func(tlsCfg *tls.Config, conns int) *http.Client + + mu sync.Mutex + // clients holds one set of clients per connection cap. + clients map[int]*chconn.HTTPClients + + // maxResponseBytes optionally overrides maxCHResponseBytes. Test-only + // seam for the cap-overflow path; not a production knob. + maxResponseBytes int64 +} + +func newCHReader(build func(tlsCfg *tls.Config, conns int) *http.Client) *chReader { + return &chReader{build: build, clients: map[int]*chconn.HTTPClients{}} +} + +// sharedCHReader serves both cached read handlers, so a tenant's reads share +// one connection cap whichever route they come in by, as they shared its +// native pool. +var sharedCHReader = newCHReader(readerHTTPClient) + +// readerHTTPClient is the read paths' client: net/http's default transport +// with the target's TLS config and at most conns connections per server. +// Tenants on one server with the same cap share it, as tenants on one tuple +// share a pool; a read that finds every connection busy waits for one until +// its deadline. Like the proxy's client it has no Timeout — every request +// carries a deadline — and does not chase redirects: the target is operator +// config, and ClickHouse does not redirect in normal operation. +func readerHTTPClient(tlsCfg *tls.Config, conns int) *http.Client { + transport := http.DefaultTransport.(*http.Transport).Clone() + transport.TLSClientConfig = tlsCfg + transport.MaxConnsPerHost = conns + transport.MaxIdleConnsPerHost = conns + return &http.Client{ + Transport: transport, + CheckRedirect: func(*http.Request, []*http.Request) error { + return http.ErrUseLastResponse + }, + } +} + +// client returns the client for target under a cap of conns connections. +func (c *chReader) client(target chconn.Target, conns int) *http.Client { + if conns <= 0 { + conns = defaultReadConns + } + c.mu.Lock() + clients, ok := c.clients[conns] + if !ok { + clients = chconn.NewHTTPClients(func(tlsCfg *tls.Config) *http.Client { return c.build(tlsCfg, conns) }) + c.clients[conns] = clients + } + c.mu.Unlock() + return clients.For(target) +} + +// chResponseTooLargeError is a response past the read paths' buffer cap. The +// same query overflows again, so it is not retryable. +type chResponseTooLargeError struct{ limit int64 } + +func (e *chResponseTooLargeError) Error() string { + return fmt.Sprintf("clickhouse response exceeded %d bytes; narrow the query", e.limit) +} + +// do sends req to target and returns the rows as a JSON array. A refusal is a +// *chconn.HTTPError, a failure on the way is wrapped with %w, so +// chconn.Classify can class either; an oversized response is a +// *chResponseTooLargeError. +func (c *chReader) do(ctx context.Context, target chconn.Target, conns int, req chRequest) ([]byte, error) { + u, err := url.Parse(target.URL) + if err != nil { + return nil, fmt.Errorf("invalid clickhouse endpoint: %w", err) + } + q := u.Query() + for k, v := range chReadSettingsFixed { + q.Set(k, v) + } + if !req.write { + q.Set("readonly", "2") + } + if target.Database != "" { + q.Set("database", target.Database) + } + for k, v := range req.settings { + q.Set(k, v) + } + for i, p := range req.params { + q.Set("param_p"+strconv.Itoa(i), p) + } + u.RawQuery = q.Encode() + + httpReq, err := http.NewRequestWithContext(ctx, http.MethodPost, u.String(), strings.NewReader(req.sql)) + if err != nil { + return nil, fmt.Errorf("clickhouse request: %w", err) + } + // The configured headers first, so WaveHouse's own win over a same-named + // one. + for name, value := range target.Headers { + httpReq.Header.Set(name, value) + } + httpReq.Header.Set("Content-Type", "text/plain; charset=utf-8") + if target.Username != "" { + httpReq.Header.Set("X-ClickHouse-User", target.Username) + } + if target.Password != "" { + httpReq.Header.Set("X-ClickHouse-Key", target.Password) + } + + // #nosec G704 -- the destination is operator config, not caller input: + // the scheme, host and path come from chconn.Target and only RawQuery is + // replaced, with url.Values.Encode() percent-encoding every key and value. + resp, err := c.client(target, conns).Do(httpReq) + if err != nil { + return nil, fmt.Errorf("clickhouse request failed: %w", err) + } + defer func() { _ = resp.Body.Close() }() + + respCap := int64(maxCHResponseBytes) + if c.maxResponseBytes > 0 { + respCap = c.maxResponseBytes + } + // One byte past the cap tells "exactly cap or more" without a second read. + body, err := io.ReadAll(io.LimitReader(resp.Body, respCap+1)) + if err != nil { + return nil, fmt.Errorf("read clickhouse response: %w", err) + } + if int64(len(body)) > respCap { + return nil, &chResponseTooLargeError{limit: respCap} + } + if resp.StatusCode != http.StatusOK { + return nil, chconn.NewHTTPError(&http.Response{StatusCode: resp.StatusCode, Header: resp.Header, Body: io.NopCloser(bytes.NewReader(body))}) + } + rows, tail := jsonEachRowArray(body) + if tail != nil { + // Text after the rows is an exception ClickHouse appended once it + // could no longer revise the 200 — a setting above overridden on the + // way. It is the error it would have been had it arrived in time. + if i := bytes.Index(tail, []byte("Code: ")); i > 0 { + tail = tail[i:] + } + return nil, chconn.NewHTTPError(&http.Response{StatusCode: resp.StatusCode, Header: resp.Header, Body: io.NopCloser(bytes.NewReader(tail))}) + } + return rows, nil +} + +// jsonEachRowArray frames ClickHouse's newline-delimited JSONEachRow output as +// the JSON array the read endpoints return, copying the rows through +// untouched. A JSONEachRow row is one JSON object on one line — a newline in +// a string is escaped — so splitting on '\n' cannot cut a row in half. No +// rows is an empty body and becomes `[]`, never `null`: SDK consumers do +// `data.length` on every response. From the first line that does not open an +// object, the rest of the body is returned as tail. +func jsonEachRowArray(body []byte) (rows, tail []byte) { + out := make([]byte, 0, len(body)+2) + out = append(out, '[') + first := true + for len(body) > 0 { + line, rest, _ := bytes.Cut(body, []byte{'\n'}) + if len(line) == 0 { + body = rest + continue + } + if line[0] != '{' { + return nil, body + } + if !first { + out = append(out, ',') + } + out = append(out, line...) + first = false + body = rest + } + return append(out, ']'), nil +} + +// checkParamSizes refuses bound values the HTTP interface would refuse: each +// one past ClickHouse's per-field limit, or all of them past what the request +// line holds. A long `in` list is the usual cause, and the caller is the one +// who can split it. +func checkParamSizes(params []string) error { + total := 0 + for _, p := range params { + n := len(url.QueryEscape(p)) + if n > chMaxFieldBytes { + return fmt.Errorf("filter value too large: %d bytes once encoded, over the %d ClickHouse's HTTP interface takes; split a long in list across queries", n, chMaxFieldBytes) + } + total += n + } + if total > chMaxURIBytes-chURIHeadroom { + return fmt.Errorf("filter values too large: %d bytes once encoded, over the %d ClickHouse's HTTP interface takes; split long in lists across queries", total, chMaxURIBytes-chURIHeadroom) + } + return nil +} + +// targetOf is target's answer for store — the tenant's HTTP wiring — and the +// zero Target, a tenant on no pool, for an unwired source. +func targetOf(target func(*settings.Store) chconn.Target, store *settings.Store) chconn.Target { + if target == nil { + return chconn.Target{} + } + return target(store) +} + +// connsOf is conns's answer for store, zero (the default cap) for an unwired +// source. +func connsOf(conns func(*settings.Store) int, store *settings.Store) int { + if conns == nil { + return 0 + } + return conns(store) +} + +// timeoutOf is timeout's answer for store, zero for an unwired source. +func timeoutOf(timeout func(*settings.Store) time.Duration, store *settings.Store) time.Duration { + if timeout == nil { + return 0 + } + return timeout(store) +} diff --git a/internal/api/clickhouse_http_test.go b/internal/api/clickhouse_http_test.go new file mode 100644 index 00000000..66b64649 --- /dev/null +++ b/internal/api/clickhouse_http_test.go @@ -0,0 +1,404 @@ +package api + +import ( + "context" + "crypto/tls" + "errors" + "fmt" + "io" + "net/http" + "net/http/httptest" + "net/url" + "strconv" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/chconn" + "github.com/Wave-RF/WaveHouse/internal/settings" +) + +// fakeCHURL is the address a fakeCH target names; nothing listens on it. +const fakeCHURL = "http://clickhouse.test:8123" + +// fakeCH is ClickHouse's HTTP interface in memory: an http.RoundTripper that +// records every request and hands it to answer, which writes the response (a +// 200 with no rows when nil). In memory rather than on a socket so a test +// under synctest can see a request held inside it. +type fakeCH struct { + answer func(w http.ResponseWriter, req *chSeen) + // err, when set, is the transport failure every request meets instead. + err error + + mu sync.Mutex + seen []*chSeen + // reads and writes count requests sent with and without readonly=2. + reads, writes atomic.Int32 +} + +// chSeen is one request a fakeCH received. +type chSeen struct { + sql string + query url.Values + header http.Header +} + +func (f *fakeCH) RoundTrip(r *http.Request) (*http.Response, error) { + body, _ := io.ReadAll(r.Body) + _ = r.Body.Close() + seen := &chSeen{sql: string(body), query: r.URL.Query(), header: r.Header.Clone()} + f.mu.Lock() + f.seen = append(f.seen, seen) + f.mu.Unlock() + if seen.query.Get("readonly") == "2" { + f.reads.Add(1) + } else { + f.writes.Add(1) + } + if f.err != nil { + return nil, f.err + } + rec := httptest.NewRecorder() + if f.answer != nil { + f.answer(rec, seen) + } + resp := rec.Result() + resp.Request = r + return resp, nil +} + +// reader is a chReader whose every client goes through f. +func (f *fakeCH) reader() *chReader { + return newCHReader(func(*tls.Config, int) *http.Client { return &http.Client{Transport: f} }) +} + +// target is a Target getter naming f, whatever the tenant. +func (f *fakeCH) target(*settings.Store) chconn.Target { return chconn.Target{URL: fakeCHURL} } + +// last is the most recent request f received, nil before the first. +func (f *fakeCH) last() *chSeen { + f.mu.Lock() + defer f.mu.Unlock() + if len(f.seen) == 0 { + return nil + } + return f.seen[len(f.seen)-1] +} + +// sql is the statement of the most recent request, "" before the first — +// which is itself the assertion for a request refused before ClickHouse. +func (f *fakeCH) sql() string { + if s := f.last(); s != nil { + return s.sql + } + return "" +} + +// params are the most recent request's bound values in placeholder order: +// what {p0:…}, {p1:…}, … received. +func (f *fakeCH) params() []string { + s := f.last() + if s == nil { + return nil + } + var out []string + for i := 0; ; i++ { + v, ok := s.query["param_p"+strconv.Itoa(i)] + if !ok { + return out + } + out = append(out, v[0]) + } +} + +// setting is one query-string setting of the most recent request. +func (f *fakeCH) setting(name string) string { + if s := f.last(); s != nil { + return s.query.Get(name) + } + return "" +} + +// answerRows answers with body as ClickHouse's JSONEachRow output. +func answerRows(body string) func(http.ResponseWriter, *chSeen) { + return func(w http.ResponseWriter, _ *chSeen) { _, _ = io.WriteString(w, body) } +} + +// answerException answers as ClickHouse refusing a statement: status, the +// exception code header (none for 0), and body. +func answerException(status int, code int32, body string) func(http.ResponseWriter, *chSeen) { + return func(w http.ResponseWriter, _ *chSeen) { + if code > 0 { + w.Header().Set("X-ClickHouse-Exception-Code", strconv.Itoa(int(code))) + } + w.WriteHeader(status) + _, _ = io.WriteString(w, body) + } +} + +// read is a read of sql against f's target with no params or settings. +func read(t *testing.T, r *chReader, sql string) ([]byte, error) { + t.Helper() + return r.do(t.Context(), chconn.Target{URL: fakeCHURL}, 0, chRequest{sql: sql}) +} + +// TestCHReader_ResponseShape pins the body the cached read paths serve. +// ClickHouse's JSONEachRow output is newline-delimited, the endpoints' +// contract is a JSON array, and a zero-row result must be `[]` and never +// `null` — every SDK consumer does `data.length` on it. +func TestCHReader_ResponseShape(t *testing.T) { + t.Parallel() + tests := []struct { + name string + body string + want string + }{ + {"no rows is an empty body", "", "[]"}, + {"one row", "{\"a\":1}\n", `[{"a":1}]`}, + {"several rows", "{\"a\":1}\n{\"a\":2}\n{\"a\":3}\n", `[{"a":1},{"a":2},{"a":3}]`}, + {"no trailing newline", "{\"a\":1}", `[{"a":1}]`}, + {"blank lines are skipped", "\n{\"a\":1}\n\n", `[{"a":1}]`}, + { + // A newline inside a value is escaped by JSONEachRow, so splitting + // on '\n' cannot cut a row in half. + name: "an escaped newline inside a value is not a row boundary", + body: "{\"a\":\"x\\ny\"}\n", + want: `[{"a":"x\ny"}]`, + }, + { + // The bytes are ClickHouse's: key order, a decimal's digits and a + // timestamp's spelling all pass through untouched. + name: "rows are copied, not re-encoded", + body: "{\"z\":1,\"a\":12.50,\"ts\":\"2026-01-15 10:30:00.120\"}\n", + want: `[{"z":1,"a":12.50,"ts":"2026-01-15 10:30:00.120"}]`, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + got, err := read(t, (&fakeCH{answer: answerRows(tt.body)}).reader(), "SELECT 1") + require.NoError(t, err) + assert.Equal(t, tt.want, string(got)) + }) + } +} + +// TestCHReader_Request pins what reaches ClickHouse: the SQL as the POST body, +// the fixed rendering and failure settings, readonly=2 on a read and not on a +// write, the database, the bound values as param_pN in placeholder order, the +// per-query settings, and the credentials as ClickHouse's own headers — over +// the target's configured headers, which ride along. +func TestCHReader_Request(t *testing.T) { + t.Parallel() + ch := &fakeCH{} + target := chconn.Target{ + URL: fakeCHURL, Username: "u", Password: "p", Database: "warehouse", + Headers: map[string]string{"X-Proxy-Token": "t", "X-ClickHouse-User": "spoofed"}, + } + _, err := ch.reader().do(t.Context(), target, 0, chRequest{ + sql: "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` IN {p1:Array(String)}", + params: []string{"/home", "['x','y']"}, + settings: map[string]string{"max_rows_to_read": "1", "read_overflow_mode": "throw"}, + }) + require.NoError(t, err) + + got := ch.last() + assert.Equal(t, "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` IN {p1:Array(String)}", got.sql) + for name, want := range chReadSettingsFixed { + assert.Equal(t, want, got.query.Get(name), name) + } + assert.Equal(t, "2", got.query.Get("readonly")) + assert.Equal(t, "warehouse", got.query.Get("database")) + assert.Equal(t, []string{"/home", "['x','y']"}, ch.params()) + assert.Equal(t, "1", got.query.Get("max_rows_to_read")) + assert.Equal(t, "throw", got.query.Get("read_overflow_mode")) + assert.Equal(t, "u", got.header.Get("X-ClickHouse-User")) + assert.Equal(t, "p", got.header.Get("X-ClickHouse-Key")) + assert.Equal(t, "t", got.header.Get("X-Proxy-Token")) + + _, err = ch.reader().do(t.Context(), target, 0, chRequest{sql: "INSERT INTO t VALUES (1)", write: true}) + require.NoError(t, err) + assert.False(t, ch.last().query.Has("readonly"), "a write must not be sent read-only") + assert.Equal(t, int32(1), ch.reads.Load()) + assert.Equal(t, int32(1), ch.writes.Load()) +} + +// TestCHReader_Errors covers every way a read fails, each typed so +// chconn.Classify and chFailureOf can answer it by class, and each carrying +// ClickHouse's own message — the only diagnostic an operator gets. +func TestCHReader_Errors(t *testing.T) { + t.Parallel() + + t.Run("a refusal carries its code from the header", func(t *testing.T) { + t.Parallel() + _, err := read(t, (&fakeCH{answer: answerException(500, 158, "Code: 158. DB::Exception: Limit for rows exceeded. (TOO_MANY_ROWS)\n")}).reader(), "SELECT 1") + he, ok := errors.AsType[*chconn.HTTPError](err) + require.True(t, ok, "%v", err) + assert.Equal(t, int32(158), he.Code) + assert.Equal(t, 500, he.StatusCode) + assert.Contains(t, chErrorMessage(err), "TOO_MANY_ROWS") + assert.NotContains(t, chErrorMessage(err), "HTTP 500", "the caller reads ClickHouse's text, not the wrapper's") + }) + + t.Run("or from the body when the header is missing", func(t *testing.T) { + t.Parallel() + _, err := read(t, (&fakeCH{answer: answerException(500, 0, "Code: 62. DB::Exception: Syntax error")}).reader(), "SELEC 1") + code, ok := chconn.ExceptionCode(err) + require.True(t, ok, "%v", err) + assert.Equal(t, int32(62), code) + }) + + t.Run("an empty refusal still names the status", func(t *testing.T) { + t.Parallel() + _, err := read(t, (&fakeCH{answer: answerException(502, 0, "")}).reader(), "SELECT 1") + status, ok := chconn.HTTPStatus(err) + require.True(t, ok, "%v", err) + assert.Equal(t, 502, status) + assert.Contains(t, chErrorMessage(err), "502") + }) + + t.Run("an exception appended after the rows is an error, not a row", func(t *testing.T) { + t.Parallel() + // What ClickHouse sends when a query fails after the 200 went out — + // which wait_end_of_query exists to prevent, unless something on the + // way overrides it. Without the check the text would be spliced into + // the array as if it were data. + body := "{\"a\":1}\nCode: 241. DB::Exception: Memory limit (total) exceeded. (MEMORY_LIMIT_EXCEEDED)\n" + _, err := read(t, (&fakeCH{answer: answerRows(body)}).reader(), "SELECT 1") + code, ok := chconn.ExceptionCode(err) + require.True(t, ok, "%v", err) + assert.Equal(t, int32(241), code) + assert.True(t, strings.HasPrefix(chErrorMessage(err), "Code: 241."), chErrorMessage(err)) + }) + + t.Run("trailing text with no code has no verdict", func(t *testing.T) { + t.Parallel() + _, err := read(t, (&fakeCH{answer: answerRows("{\"a\":1}\nsomething odd\n")}).reader(), "SELECT 1") + require.Error(t, err) + _, hasCode := chconn.ExceptionCode(err) + assert.False(t, hasCode) + assert.Equal(t, chconn.Unknown, chconn.Classify(err)) + }) + + t.Run("an oversized response is refused, not buffered", func(t *testing.T) { + t.Parallel() + r := (&fakeCH{answer: answerRows(strings.Repeat("{\"a\":1}\n", 100))}).reader() + r.maxResponseBytes = 32 + _, err := read(t, r, "SELECT 1") + tooLarge, ok := errors.AsType[*chResponseTooLargeError](err) + require.True(t, ok, "%v", err) + assert.Equal(t, int64(32), tooLarge.limit) + assert.Contains(t, err.Error(), "exceeded 32 bytes") + }) + + t.Run("an unreachable endpoint is unavailable", func(t *testing.T) { + t.Parallel() + r := newCHReader(readerHTTPClient) + _, err := r.do(t.Context(), chconn.Target{URL: "http://" + closedAddr(t)}, 0, chRequest{sql: "SELECT 1"}) + require.Error(t, err) + assert.Contains(t, err.Error(), "clickhouse request failed") + assert.Equal(t, chconn.Unavailable, chconn.Classify(err)) + }) + + t.Run("an endpoint that does not parse", func(t *testing.T) { + t.Parallel() + _, err := (&fakeCH{}).reader().do(t.Context(), chconn.Target{URL: "http://[::1"}, 0, chRequest{sql: "SELECT 1"}) + require.Error(t, err) + assert.Contains(t, err.Error(), "invalid clickhouse endpoint") + }) +} + +// TestCHReader_ClientsPerCap: a tenant's reads share one transport per +// connection cap, capped at that many connections per server; no cap is +// defaultReadConns, and two caps never share a transport. +func TestCHReader_ClientsPerCap(t *testing.T) { + t.Parallel() + r := newCHReader(readerHTTPClient) + target := chconn.Target{URL: fakeCHURL} + maxConns := func(c *http.Client) int { return c.Transport.(*http.Transport).MaxConnsPerHost } + + assert.Equal(t, defaultReadConns, maxConns(r.client(target, 0))) + assert.Same(t, r.client(target, 0), r.client(target, defaultReadConns)) + assert.Equal(t, 7, maxConns(r.client(target, 7))) + assert.Same(t, r.client(target, 7), r.client(target, 7)) + assert.NotSame(t, r.client(target, 7), r.client(target, 8)) + assert.Equal(t, 7, r.client(target, 7).Transport.(*http.Transport).MaxIdleConnsPerHost) +} + +// TestCHReader_ConnectionCapHolds: past the cap, a read waits for a +// connection rather than opening another. The second read is in flight while +// the first holds the only connection; it must not reach the server until the +// first is done. +func TestCHReader_ConnectionCapHolds(t *testing.T) { + t.Parallel() + var mu sync.Mutex + active, peak := 0, 0 + entered := make(chan struct{}, 2) + release := make(chan struct{}) + srv := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) { + mu.Lock() + active++ + peak = max(peak, active) + mu.Unlock() + entered <- struct{}{} + <-release + mu.Lock() + active-- + mu.Unlock() + })) + t.Cleanup(srv.Close) + + r := newCHReader(readerHTTPClient) + target := chconn.Target{URL: srv.URL} + var wg sync.WaitGroup + for range 2 { + wg.Go(func() { + _, err := r.do(context.Background(), target, 1, chRequest{sql: "SELECT 1"}) + assert.NoError(t, err) + }) + } + <-entered + select { + case <-entered: + t.Fatal("a second connection opened past a cap of 1") + case <-time.After(100 * time.Millisecond): + } + close(release) + wg.Wait() + mu.Lock() + defer mu.Unlock() + assert.Equal(t, 1, peak) +} + +// TestCheckParamSizes: a bound value is refused when its percent-encoded form +// passes ClickHouse's per-field limit — measured to the byte, 131072 read and +// 132096 refused — and all of them together when they pass what the request +// line holds. +func TestCheckParamSizes(t *testing.T) { + t.Parallel() + assert.NoError(t, checkParamSizes(nil)) + assert.NoError(t, checkParamSizes([]string{strings.Repeat("a", chMaxFieldBytes)})) + err := checkParamSizes([]string{strings.Repeat("a", chMaxFieldBytes+1)}) + require.Error(t, err) + assert.Contains(t, err.Error(), "in list") + // The encoded size counts: a quote is three bytes on the wire. + require.Error(t, checkParamSizes([]string{strings.Repeat("'", chMaxFieldBytes/3+1)})) + + under := strings.Repeat("a", chMaxFieldBytes) + var many []string + for range (chMaxURIBytes - chURIHeadroom) / chMaxFieldBytes { + many = append(many, under) + } + assert.NoError(t, checkParamSizes(many)) + require.Error(t, checkParamSizes(append(many, under))) +} + +// chExceptionBody is ClickHouse's own refusal text for code. +func chExceptionBody(code int32, msg string) string { + return fmt.Sprintf("Code: %d. DB::Exception: %s. (version 26.6.3.62 (official build))\n", code, msg) +} diff --git a/internal/api/pipes.go b/internal/api/pipes.go index 3a850d3d..9e3bc965 100644 --- a/internal/api/pipes.go +++ b/internal/api/pipes.go @@ -7,9 +7,9 @@ import ( "net/http" "time" - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/Wave-RF/WaveHouse/internal/auth" "github.com/Wave-RF/WaveHouse/internal/cache" + "github.com/Wave-RF/WaveHouse/internal/chconn" "github.com/Wave-RF/WaveHouse/internal/pipes" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/settings" @@ -28,11 +28,17 @@ type PipesHandler struct { // is tenant-exempt, so they carry no request tenant and read the one // ?tenant= names, the default one without it (opsStore). Tenants *settings.Registry - // CHConn yields the request tenant's connection (chconn.Pools.For in - // production); nil is a tenant on no pool, a 503. - CHConn func(*settings.Store) driver.Conn - Cache cache.Cache - sf singleflight.Group + // Target yields the request tenant's ClickHouse HTTP wiring + // (chconn.Pools.Target in production); the zero Target is a tenant on no + // pool, a 503. + Target func(*settings.Store) chconn.Target + // MaxConns caps the tenant's concurrent reads + // ((*settings.Store).ClickHouse().MaxOpenConns in production); nil or + // non-positive is defaultReadConns. + MaxConns func(*settings.Store) int + Cache cache.Cache + sf singleflight.Group + ch *chReader // queryTimeout bounds each pipe execution, read per request off the // tenant's settings ((*settings.Store).ClickHouse().QueryTimeout in // production) so a settings reload applies without a restart. @@ -46,8 +52,8 @@ type PipesHandler struct { maxRequestBytes int64 } -func NewPipesHandler(source func(*settings.Store) pipes.Source, policySource PolicySource, conn func(*settings.Store) driver.Conn, c cache.Cache, queryTimeout func(*settings.Store) time.Duration) *PipesHandler { - return &PipesHandler{Source: source, PolicySource: policySource, CHConn: conn, Cache: c, queryTimeout: queryTimeout} +func NewPipesHandler(source func(*settings.Store) pipes.Source, policySource PolicySource, target func(*settings.Store) chconn.Target, c cache.Cache, queryTimeout func(*settings.Store) time.Duration) *PipesHandler { + return &PipesHandler{Source: source, PolicySource: policySource, Target: target, Cache: c, ch: sharedCHReader, queryTimeout: queryTimeout} } // List returns all named queries of the ?tenant= (admin endpoint). @@ -146,14 +152,18 @@ func (h *PipesHandler) Execute(w http.ResponseWriter, r *http.Request) { } } - sql, params, err := pipes.BindParams(q, supplied) + // BindParams inlines every value as an escaped SQL literal — a pipe's + // placeholders can sit anywhere in the statement, including positions + // (LIMIT, an identifier) where a bound parameter is not legal — so the + // rendered SQL carries no placeholders and nothing is bound here. + sql, err := pipes.BindParams(q, supplied) if err != nil { writeJSONError(w, http.StatusBadRequest, err.Error()) return } if IsMutation(sql) { - h.executeWrite(w, r, store, sql, params) + h.executeWrite(w, r, store, sql) return } @@ -166,7 +176,7 @@ func (h *PipesHandler) Execute(w http.ResponseWriter, r *http.Request) { // once its pool below is taken — orphans the fill. // TODO: once pipes expose their tables/scopes, pass them as deps here so writes // invalidate cached pipe results. - cacheKey := queryCacheKey(store.Tenant(), sql, params) + cacheKey := queryCacheKey(store.Tenant(), sql, nil) var entry cache.Entry var snap cache.Snapshot if h.Cache != nil { @@ -176,8 +186,8 @@ func (h *PipesHandler) Execute(w http.ResponseWriter, r *http.Request) { // The tenant's pool, ahead of serving a hit: a tenant on none — its // tuple could not be opened, such as by the connection ceiling — fails // closed rather than serve what it cached before (#583 story 6). - conn := connOf(h.CHConn, store) - if conn == nil { + target := targetOf(h.Target, store) + if target.URL == "" { writeUnavailable(w, noConnectionMessage, retryAfterPool) return } @@ -190,59 +200,58 @@ func (h *PipesHandler) Execute(w http.ResponseWriter, r *http.Request) { // Execute with singleflight. v, err, _ := h.sf.Do(cacheKey, func() (interface{}, error) { - data, queryDuration, err := h.run(r.Context(), store, conn, sql, params) + start := time.Now() + data, err := h.run(r.Context(), store, target, chRequest{sql: sql}) if err != nil { return nil, err } if h.Cache != nil { - _ = h.Cache.Set(r.Context(), snap, data, cache.QueryTimeToTTL(queryDuration)) + _ = h.Cache.Set(r.Context(), snap, data, cache.QueryTimeToTTL(time.Since(start))) } return data, nil }) if err != nil { - writeCHError(w, r, err, err.Error(), http.StatusInternalServerError, queryCaps{}) + writeCHError(w, r, err, chErrorMessage(err), http.StatusInternalServerError, queryCaps{readonly: true}) return } w.Header().Set("Content-Type", "application/json") w.Header().Set("X-Cache", "MISS") - _, _ = w.Write(v.([]byte)) //nolint:gosec // G705: the tenant id on the key only selects the entry; the bytes are JSON the handler marshalled from ClickHouse rows + _, _ = w.Write(v.([]byte)) //nolint:gosec // G705: the tenant id on the key only selects the entry; the bytes are ClickHouse's JSONEachRow rows framed as an array } // executeWrite runs a pipe that writes, on every call: a cached or coalesced // response would answer a repeat without executing it, silently dropping the -// write (#386) — on every instance once the cache is shared. IsMutation is the -// classifier executeCHQuery routes Exec by, so what bypasses here is exactly -// what runs as a write. no-store keeps an HTTP cache in front of a GET from -// answering a repeat the same way. -func (h *PipesHandler) executeWrite(w http.ResponseWriter, r *http.Request, store *settings.Store, sql string, params []any) { - conn := connOf(h.CHConn, store) - if conn == nil { +// write (#386) — on every instance once the cache is shared. It is the one +// statement sent without readonly=2, so what bypasses here is exactly what +// may write. no-store keeps an HTTP cache in front of a GET from answering a +// repeat the same way. +func (h *PipesHandler) executeWrite(w http.ResponseWriter, r *http.Request, store *settings.Store, sql string) { + target := targetOf(h.Target, store) + if target.URL == "" { writeUnavailable(w, noConnectionMessage, retryAfterPool) return } - data, _, err := h.run(r.Context(), store, conn, sql, params) + data, err := h.run(r.Context(), store, target, chRequest{sql: sql, write: true}) if err != nil { - writeCHWriteError(w, r, err, err.Error()) + writeCHWriteError(w, r, err, chErrorMessage(err)) return } w.Header().Set("Content-Type", "application/json") w.Header().Set("X-Cache", "BYPASS") w.Header().Set("Cache-Control", "no-store") - _, _ = w.Write(data) //nolint:gosec // G705: JSON the handler marshalled from the exec result + _, _ = w.Write(data) //nolint:gosec // G705: ClickHouse's JSONEachRow rows framed as an array, [] for a write } // run executes a pipe's bound SQL under the tenant's query timeout and -// returns the rows as JSON with how long ClickHouse took. -func (h *PipesHandler) run(ctx context.Context, store *settings.Store, conn driver.Conn, sql string, params []any) ([]byte, time.Duration, error) { - queryCtx, cancel := context.WithTimeout(ctx, timeoutOf(h.queryTimeout, store)) +// returns ClickHouse's rows as a JSON array. A pipe carries no per-role +// resource caps (allowed_roles is its whole policy), so the query timeout is +// the only limit it sends; the deadline outlasts it by capBackstop, as on +// the structured query. +func (h *PipesHandler) run(ctx context.Context, store *settings.Store, target chconn.Target, req chRequest) ([]byte, error) { + timeout := timeoutOf(h.queryTimeout, store) + queryCtx, cancel := context.WithTimeout(ctx, timeout+capBackstop) defer cancel() - start := time.Now() - rows, err := executeCHQuery(queryCtx, conn, sql, params) - queryDuration := time.Since(start) - if err != nil { - return nil, 0, err - } - data, err := json.Marshal(rows) - return data, queryDuration, err + req.settings = chReadSettings(chQueryLimits{ExecutionTime: timeout}) + return h.ch.do(queryCtx, target, connsOf(h.MaxConns, store), req) } diff --git a/internal/api/pipes_test.go b/internal/api/pipes_test.go index f3a163d4..036f0cc9 100644 --- a/internal/api/pipes_test.go +++ b/internal/api/pipes_test.go @@ -8,12 +8,10 @@ import ( "net/http/httptest" "strings" "sync" - "sync/atomic" "testing" "testing/synctest" "time" - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/Wave-RF/WaveHouse/internal/auth" "github.com/Wave-RF/WaveHouse/internal/cache" "github.com/Wave-RF/WaveHouse/internal/pipes" @@ -452,7 +450,8 @@ func TestPipesHandler_Execute_NoAllowedRoles_NonAdminDenied(t *testing.T) { } // TestPipesHandler_Execute_ArrayParamBinds: an array body param renders into an -// IN list and passes binding (failing only later at the nil ClickHouse conn). +// IN list and passes binding (failing only later at the unwired ClickHouse +// target). func TestPipesHandler_Execute_ArrayParamBinds(t *testing.T) { t.Parallel() store := staticPipes( @@ -471,7 +470,8 @@ func TestPipesHandler_Execute_ArrayParamBinds(t *testing.T) { safeHandle(h.Execute, w, withTenant(r)) - // Binding succeeded — the only failure left is the nil conn, never a 400. + // Binding succeeded — the only failure left is the unwired target, never a + // 400. assert.NotEqual(t, http.StatusBadRequest, w.Code) assert.NotEqual(t, http.StatusNotFound, w.Code) } @@ -517,30 +517,6 @@ func TestPipesHandler_Execute_NoAllowedRoles_AdminAllowed(t *testing.T) { assert.NotEqual(t, http.StatusNotFound, w.Code) } -// writeConn counts Exec and Query calls, and every Exec returns err. With -// gate set, every Exec reports itself on entered and holds until gate is -// closed, so a test can hold requests in flight together. -type writeConn struct { - driver.Conn - execs, queries atomic.Int32 - entered, gate chan struct{} - err error -} - -func (c *writeConn) Exec(context.Context, string, ...any) error { - c.execs.Add(1) - if c.gate != nil { - c.entered <- struct{}{} - <-c.gate - } - return c.err -} - -func (c *writeConn) Query(context.Context, string, ...any) (driver.Rows, error) { - c.queries.Add(1) - return &chainEmptyRows{}, nil -} - // pipeCallAs runs the pipe name as the writer role and returns the recorder. func pipeCallAs(t *testing.T, h *PipesHandler, name string) *httptest.ResponseRecorder { t.Helper() @@ -550,13 +526,16 @@ func pipeCallAs(t *testing.T, h *PipesHandler, name string) *httptest.ResponseRe return w } -func writerPipesHandler(t *testing.T, conn driver.Conn, c cache.Cache, queries ...*pipes.NamedQuery) *PipesHandler { +// writerPipesHandler serves queries to the writer role, through ch. +func writerPipesHandler(t *testing.T, ch *fakeCH, c cache.Cache, queries ...*pipes.NamedQuery) *PipesHandler { t.Helper() for _, q := range queries { q.AllowedRoles = []string{"writer"} } timeout := func(*settings.Store) time.Duration { return 5 * time.Second } - return NewPipesHandler(staticPipes(queries...), staticPolicy(&policy.Policy{}), fixedConn(conn), c, timeout) + h := NewPipesHandler(staticPipes(queries...), staticPolicy(&policy.Policy{}), ch.target, c, timeout) + h.ch = ch.reader() + return h } // #386: a pipe that writes executes on every call. Served from the cache, a @@ -574,8 +553,8 @@ func TestPipesHandler_Execute_MutationRunsEveryCall(t *testing.T) { l1, err := cache.NewLocal(1 << 20) require.NoError(t, err) t.Cleanup(func() { _ = l1.Close() }) - conn := &writeConn{} - h := writerPipesHandler(t, conn, l1, &pipes.NamedQuery{Name: "log", SQL: sql}) + ch := &fakeCH{} + h := writerPipesHandler(t, ch, l1, &pipes.NamedQuery{Name: "log", SQL: sql}) for range 3 { w := pipeCallAs(t, h, "log") @@ -585,8 +564,8 @@ func TestPipesHandler_Execute_MutationRunsEveryCall(t *testing.T) { assert.JSONEq(t, `[]`, w.Body.String()) l1.Wait() } - assert.Equal(t, int32(3), conn.execs.Load(), "every call must reach ClickHouse") - assert.Zero(t, conn.queries.Load()) + assert.Equal(t, int32(3), ch.writes.Load(), "every call must reach ClickHouse, not read-only") + assert.Zero(t, ch.reads.Load()) }) } } @@ -597,8 +576,9 @@ func TestPipesHandler_Execute_MutationRunsEveryCall(t *testing.T) { func TestPipesHandler_Execute_ConcurrentMutationsNotCoalesced(t *testing.T) { synctest.Test(t, func(t *testing.T) { const calls = 3 - conn := &writeConn{entered: make(chan struct{}, calls), gate: make(chan struct{})} - h := writerPipesHandler(t, conn, nil, &pipes.NamedQuery{Name: "log", SQL: "INSERT INTO audit_log VALUES ({{msg}}, now())"}) + entered, gate := make(chan struct{}, calls), make(chan struct{}) + ch := gatedCH(entered, gate) + h := writerPipesHandler(t, ch, nil, &pipes.NamedQuery{Name: "log", SQL: "INSERT INTO audit_log VALUES ({{msg}}, now())"}) var wg sync.WaitGroup for range calls { wg.Go(func() { @@ -607,10 +587,10 @@ func TestPipesHandler_Execute_ConcurrentMutationsNotCoalesced(t *testing.T) { }) } synctest.Wait() - assert.Len(t, conn.entered, calls, "writes in flight once every request is blocked") - close(conn.gate) + assert.Len(t, entered, calls, "writes in flight once every request is blocked") + close(gate) wg.Wait() - assert.Equal(t, int32(calls), conn.execs.Load()) + assert.Equal(t, int32(calls), ch.writes.Load()) }) } @@ -621,8 +601,8 @@ func TestPipesHandler_Execute_ReadPipeStaysCached(t *testing.T) { l1, err := cache.NewLocal(1 << 20) require.NoError(t, err) t.Cleanup(func() { _ = l1.Close() }) - conn := &writeConn{} - h := writerPipesHandler(t, conn, l1, &pipes.NamedQuery{Name: "recent", SQL: "SELECT * FROM insert_log WHERE msg = {{msg}}"}) + ch := &fakeCH{} + h := writerPipesHandler(t, ch, l1, &pipes.NamedQuery{Name: "recent", SQL: "SELECT * FROM insert_log WHERE msg = {{msg}}"}) for _, want := range []string{"MISS", "HIT", "HIT"} { w := pipeCallAs(t, h, "recent") @@ -630,6 +610,40 @@ func TestPipesHandler_Execute_ReadPipeStaysCached(t *testing.T) { assert.Equal(t, want, w.Header().Get("X-Cache")) l1.Wait() } - assert.Equal(t, int32(1), conn.queries.Load()) - assert.Zero(t, conn.execs.Load()) + assert.Equal(t, int32(1), ch.reads.Load()) + assert.Zero(t, ch.writes.Load()) +} + +// What a pipe sends ClickHouse: the bound SQL as the body, its query timeout +// as max_execution_time, and readonly=2 on a read — the statement IsMutation +// reads as a write is the one sent without it. +func TestPipesHandler_Execute_RequestSettings(t *testing.T) { + t.Parallel() + ch := &fakeCH{} + h := writerPipesHandler(t, ch, nil, + &pipes.NamedQuery{Name: "recent", SQL: "SELECT * FROM log WHERE msg = {{msg}}"}, + &pipes.NamedQuery{Name: "log", SQL: "INSERT INTO log VALUES ({{msg}})"}, + ) + + require.Equal(t, http.StatusOK, pipeCallAs(t, h, "recent").Code) + assert.Equal(t, "SELECT * FROM log WHERE msg = 'hello'", ch.sql()) + assert.Equal(t, "2", ch.setting("readonly")) + assert.Equal(t, "5", ch.setting("max_execution_time")) + assert.Empty(t, ch.setting("max_result_rows"), "a pipe carries no role caps") + + require.Equal(t, http.StatusOK, pipeCallAs(t, h, "log").Code) + assert.Equal(t, "INSERT INTO log VALUES ('hello')", ch.sql()) + assert.False(t, ch.last().query.Has("readonly")) + assert.Equal(t, "5", ch.setting("max_execution_time")) +} + +// TestPipesHandler_Execute_ServesClickHouseBytes: a pipe's response is +// ClickHouse's own rendering, rows framed as an array. +func TestPipesHandler_Execute_ServesClickHouseBytes(t *testing.T) { + t.Parallel() + ch := &fakeCH{answer: answerRows("{\"page\":\"/a\",\"n\":1}\n{\"page\":\"/b\",\"n\":2}\n")} + h := writerPipesHandler(t, ch, nil, &pipes.NamedQuery{Name: "top", SQL: "SELECT page, n FROM t"}) + w := pipeCallAs(t, h, "top") + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + assert.Equal(t, `[{"page":"/a","n":1},{"page":"/b","n":2}]`, w.Body.String()) } diff --git a/internal/api/clickhouse_exec.go b/internal/api/sql_classify.go similarity index 68% rename from internal/api/clickhouse_exec.go rename to internal/api/sql_classify.go index 51d41e58..a5c6dbe5 100644 --- a/internal/api/clickhouse_exec.go +++ b/internal/api/sql_classify.go @@ -1,97 +1,16 @@ package api import ( - "context" - "fmt" - "reflect" "strings" - "time" "unicode/utf8" - - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" - "github.com/Wave-RF/WaveHouse/internal/settings" - "github.com/google/uuid" ) -// connOf is conn's answer for store — the tenant's pool — and nil for an -// unwired source or a tenant on no pool. The nil is untyped: a nil *Manager -// inside a non-nil driver.Conn would pass a nil check and panic on use. -func connOf(conn func(*settings.Store) driver.Conn, store *settings.Store) driver.Conn { - if conn == nil { - return nil - } - return conn(store) -} - -// timeoutOf is timeout's answer for store, zero for an unwired source. -func timeoutOf(timeout func(*settings.Store) time.Duration, store *settings.Store) time.Duration { - if timeout == nil { - return 0 - } - return timeout(store) -} - -// executeCHQuery runs sql against the native-protocol driver conn, -// classifying by leading SQL verb to pick the Exec-vs-Query path — -// clickhouse-go's driver.Query() errors on statements that return no -// result set, so the dispatch is correctness, not optimisation. Returns -// a row-array suitable for JSON marshalling; mutations marshal to `[]`, -// preserving the "always-an-array" response shape callers depend on. -// -// Used by the structured-query and pipes handlers — those are the cached -// read paths that need explicit Query/Exec dispatch and per-row scanning. -// The raw-SQL endpoint (/v1/ops/query) proxies straight to ClickHouse -// over HTTP and never calls this; see internal/api/query.go. -func executeCHQuery(ctx context.Context, conn driver.Conn, sql string, params []any) ([]map[string]any, error) { - if IsMutation(sql) { - if err := conn.Exec(ctx, sql, params...); err != nil { - return nil, fmt.Errorf("clickhouse exec: %w", err) - } - return []map[string]any{}, nil - } - - rows, err := conn.Query(ctx, sql, params...) - if err != nil { - return nil, fmt.Errorf("clickhouse query: %w", err) - } - defer func() { _ = rows.Close() }() - - columns := rows.ColumnTypes() - // Initialize as empty (not nil) so a zero-row result marshals to `[]`, - // not `null`. SDK consumers do `data!.length` on the response; a `null` - // crashes the client on every empty fetch. - results := []map[string]any{} - - for rows.Next() { - valPtrs := make([]any, len(columns)) - for i, col := range columns { - valPtrs[i] = reflect.New(col.ScanType()).Interface() - } - if err := rows.Scan(valPtrs...); err != nil { - return nil, fmt.Errorf("scan clickhouse row: %w", err) - } - row := make(map[string]any) - for i, col := range columns { - row[col.Name()] = reflect.ValueOf(valPtrs[i]).Elem().Interface() - } - results = append(results, transformRow(row)) - } - // rows.Next() returns false both when iteration completes successfully - // AND when the driver hits an error mid-stream (network drop, decode - // failure on a row past the first). Without this check, a partial - // result set silently masquerades as a complete one. - if err := rows.Err(); err != nil { - return nil, fmt.Errorf("iterate clickhouse rows: %w", err) - } - return results, nil -} - // mutationVerbs is a set of SQL leading keywords that don't return a result -// set — anything that mutates schema or data. Routed through Exec rather -// than Query (see executeCHQuery). Sourced from the ClickHouse statement +// set — anything that mutates schema or data. A pipe led by one runs as a +// write (PipesHandler.executeWrite). Sourced from the ClickHouse statement // reference: DML, DDL, role/privilege management, and runtime control // (SYSTEM/KILL/SET). Read-only verbs (SELECT/WITH/SHOW/DESCRIBE/EXPLAIN/ -// EXISTS) intentionally fall through to the default Query path. +// EXISTS) intentionally fall through to the read path. var mutationVerbs = map[string]struct{}{ "INSERT": {}, "UPDATE": {}, @@ -114,16 +33,17 @@ var mutationVerbs = map[string]struct{}{ "SYSTEM": {}, } -// IsMutation reports whether sql's leading statement is a non-SELECT — i.e. -// one that returns no result set and must go through Exec, not Query. -// Leading whitespace and comments are skipped as ClickHouse's lexer skips -// them, then the first bareword is matched whole, case-insensitively, against -// mutationVerbs. After a WITH list ClickHouse parses only SELECT, a FROM-first -// SELECT or INSERT INTO, so a WITH-led statement is a write exactly when it -// holds INSERT INTO at the top level (hasTopLevelInsertInto). An -// `EXECUTE AS ` prefix is looked through to the statement it runs. A -// write classified as a read goes through Query, which runs it and then fails -// the call, so a client that retries the error writes again. +// IsMutation reports whether sql's leading statement is a non-SELECT — one +// that returns no result set and may change something, so a pipe running it +// is never cached or coalesced (#386). Leading whitespace and comments are +// skipped as ClickHouse's lexer skips them, then the first bareword is matched +// whole, case-insensitively, against mutationVerbs. After a WITH list +// ClickHouse parses only SELECT, a FROM-first SELECT or INSERT INTO, so a +// WITH-led statement is a write exactly when it holds INSERT INTO at the top +// level (hasTopLevelInsertInto). An `EXECUTE AS ` prefix is looked +// through to the statement it runs. A write classified as a read runs under +// readonly=2, so ClickHouse refuses it rather than running it and having its +// empty result cached. func IsMutation(sql string) bool { s := stripLeadingSQLComments(sql) if rest, ok := skipExecuteAs(s); ok { @@ -431,23 +351,3 @@ func skipQuoted(s string, i int) int { } return len(s) } - -// transformRow converts ClickHouse-specific types to JSON-friendly values. -func transformRow(row map[string]any) map[string]any { - for k, v := range row { - switch val := v.(type) { - case uuid.UUID: - row[k] = val.String() - case [16]byte: - row[k] = uuid.UUID(val).String() - case time.Time: - row[k] = val.UTC().Format(time.RFC3339Nano) - case *time.Time: - // Nullable(DateTime…) scans as a pointer; NULL stays nil (JSON null). - if val != nil { - row[k] = val.UTC().Format(time.RFC3339Nano) - } - } - } - return row -} diff --git a/internal/api/sql_classify_test.go b/internal/api/sql_classify_test.go new file mode 100644 index 00000000..0af8f57d --- /dev/null +++ b/internal/api/sql_classify_test.go @@ -0,0 +1,40 @@ +package api + +import ( + "testing" + + "github.com/Wave-RF/WaveHouse/internal/testutil/mutationtest" + "github.com/stretchr/testify/assert" +) + +// TestIsMutation runs the shared cases; the integration suite checks the +// same cases against ClickHouse's parser. +func TestIsMutation(t *testing.T) { + t.Parallel() + for _, tc := range mutationtest.Cases { + t.Run(tc.Name, func(t *testing.T) { + t.Parallel() + assert.Equal(t, tc.Mutation, IsMutation(tc.SQL)) + }) + } +} + +// TestIsMutation_ClickHouseWhitespace pins every character ClickHouse 26.6's +// lexer accepts as whitespace, each checked against a live server: ahead of a +// write it must not hide the verb, and ahead of a read it must not make one. +func TestIsMutation_ClickHouseWhitespace(t *testing.T) { + t.Parallel() + spaces := []rune{' ', '\t', '\n', '\v', '\f', '\r', 0x85, 0xA0, 0x180E, 0x2028, 0x2029, 0x202F, 0x205F, 0x2060, 0x3000, 0xFEFF} + for r := rune(0x2000); r <= 0x200D; r++ { + spaces = append(spaces, r) + } + for _, r := range spaces { + ws := string(r) + assert.True(t, IsMutation(ws+"INSERT INTO t VALUES (1)"), "U+%04X before INSERT", r) + assert.False(t, IsMutation(ws+"SELECT 1"), "U+%04X before SELECT", r) + assert.True(t, IsMutation("WITH x AS (SELECT 1)"+ws+"INSERT INTO t SELECT * FROM x"), "U+%04X before a WITH's INSERT", r) + assert.True(t, IsMutation("WITH 1 AS x INSERT"+ws+"INTO t SELECT x"), "U+%04X between a WITH's INSERT and INTO", r) + } + // Not whitespace to ClickHouse (it rejects the statement), so not skipped. + assert.False(t, IsMutation("\u1680INSERT INTO t VALUES (1)")) +} diff --git a/internal/api/structured_query.go b/internal/api/structured_query.go index 6fb62eeb..e0d969da 100644 --- a/internal/api/structured_query.go +++ b/internal/api/structured_query.go @@ -8,10 +8,9 @@ import ( "net/http" "time" - "github.com/ClickHouse/clickhouse-go/v2" - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/Wave-RF/WaveHouse/internal/auth" "github.com/Wave-RF/WaveHouse/internal/cache" + "github.com/Wave-RF/WaveHouse/internal/chconn" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/query" "github.com/Wave-RF/WaveHouse/internal/settings" @@ -20,13 +19,19 @@ import ( // StructuredQueryHandler handles POST /v1/query?table={table} type StructuredQueryHandler struct { - // CHConn yields the request tenant's connection (chconn.Pools.For in - // production); nil is a tenant on no pool, a 503. - CHConn func(*settings.Store) driver.Conn + // Target yields the request tenant's ClickHouse HTTP wiring + // (chconn.Pools.Target in production); the zero Target is a tenant on no + // pool, a 503. + Target func(*settings.Store) chconn.Target + // MaxConns caps the tenant's concurrent reads + // ((*settings.Store).ClickHouse().MaxOpenConns in production); nil or + // non-positive is defaultReadConns. + MaxConns func(*settings.Store) int Cache cache.Cache Registry RegistrySource PolicySource PolicySource sf singleflight.Group + ch *chReader // queryTimeout bounds each query, read per request off the tenant's // settings ((*settings.Store).ClickHouse().QueryTimeout in production) // so a settings reload applies without a restart. @@ -51,7 +56,7 @@ type StructuredQueryHandler struct { } func NewStructuredQueryHandler( - conn func(*settings.Store) driver.Conn, + target func(*settings.Store) chconn.Target, c cache.Cache, registry RegistrySource, policyStore PolicySource, @@ -60,7 +65,8 @@ func NewStructuredQueryHandler( defaultMaxRows func(*settings.Store) int, ) *StructuredQueryHandler { return &StructuredQueryHandler{ - CHConn: conn, + Target: target, + ch: sharedCHReader, Cache: c, Registry: registry, PolicySource: policyStore, @@ -98,7 +104,12 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) r.Body = http.MaxBytesReader(w, r.Body, reqCap) var sq query.StructuredQuery - if err := json.NewDecoder(r.Body).Decode(&sq); err != nil { + // UseNumber so a filter value keeps the digits the caller wrote: every + // value binds as a ClickHouse String parameter, so "12.50" and an integer + // past 2^53 reach the server intact instead of through a float64. + dec := json.NewDecoder(r.Body) + dec.UseNumber() + if err := dec.Decode(&sq); err != nil { if writeMaxBytesError(w, err, reqCap) { return } @@ -159,9 +170,25 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) return } + // Bind the built SQL for ClickHouse's HTTP interface: positional `?` + // placeholders become {pN:String} / {pN:Array(String)} named parameters + // and each value becomes the text ClickHouse reads it back from. A value + // with no text form (a JSON null, an object), or one too large for the + // HTTP interface to take, is a malformed query, not a server fault. + chSQL, chParams, err := result.NamedParams() + if err == nil { + err = checkParamSizes(chParams) + } + if err != nil { + writeJSONError(w, http.StatusBadRequest, err.Error()) + return + } + // Cache key, led by the tenant the store was resolved for (#583 story 8); - // the singleflight key too. - cacheKey := queryCacheKey(store.Tenant(), result.SQL, result.Params) + // the singleflight key too. Keyed on what reaches ClickHouse, so two + // requests that differ only in a spelling the binding erases share an + // entry and two that differ in the bytes sent never do. + cacheKey := queryCacheKey(store.Tenant(), chSQL, chParams) // TODO: impl scope scope := "" @@ -183,8 +210,8 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) // The tenant's pool, ahead of serving a hit: a tenant on none — its // tuple could not be opened, such as by the connection ceiling — fails // closed rather than serve what it cached before (#583 story 6). - conn := connOf(h.CHConn, store) - if conn == nil { + target := targetOf(h.Target, store) + if target.URL == "" { writeUnavailable(w, noConnectionMessage, retryAfterPool) return } @@ -206,77 +233,53 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) // An overrun is the role's cap only when the cap is the tighter // budget; under a shorter query_timeout it reads as it does for a // role with no cap (#620). - time: timeCap > 0 && timeCap <= timeout, - memory: perms.Select.MaxMemoryUsage > 0, + time: timeCap > 0 && timeCap <= timeout, + memory: perms.Select.MaxMemoryUsage > 0, + readonly: true, } if timeCap > 0 { timeout = min(timeCap, timeout) } + // Enforce the budget server-side, not just via the request's deadline + // (#316): the settings ride on this query's URL, so they reach ClickHouse + // for this query only. The time bound goes on every read — the tighter of + // the role's cap and query_timeout — so ClickHouse stops a query nobody + // is waiting for and says which limit stopped it; the role's other caps + // only when it set them. + chSettings := chReadSettings(chQueryLimits{ + ExecutionTime: timeout, + MaxResultRows: perms.Select.MaxRows, + MaxRowsToRead: perms.Select.MaxRowsToRead, + MaxMemoryBytes: perms.Select.MaxMemoryUsage.Bytes(), + }) + conns := connsOf(h.MaxConns, store) // Execute with singleflight. v, err, _ := h.sf.Do(cacheKey, func() (interface{}, error) { - var queryCtx context.Context - var cancel context.CancelFunc - if timeCap > 0 { - // ClickHouse enforces the budget (max_execution_time below) and - // answers an overrun with TIMEOUT_EXCEEDED. A context deadline - // would let the driver raise that setting to deadline+5s and turn - // every overrun into a bare DeadlineExceeded — indistinguishable - // from a pool wait or a dial timeout, which are outages, not the - // caller's cost. - queryCtx, cancel = cancelAfter(r.Context(), timeout+capBackstop) - } else { - queryCtx, cancel = context.WithTimeout(r.Context(), timeout) - } + // The deadline outlasts max_execution_time by capBackstop, so an + // overrun comes back as ClickHouse's TIMEOUT_EXCEEDED, naming the + // limit, rather than as a dropped connection. + queryCtx, cancel := context.WithTimeout(r.Context(), timeout+capBackstop) defer cancel() - // Enforce the role's resource caps server-side, not just via the client - // context deadline (#316). The settings ride on the query context, so they - // reach ClickHouse for this query only. Server-wide backstops are - // ClickHouse's job (settings profiles / quotas); a role with no caps (e.g. - // admin) sends nothing here. An explicit max_execution_time is sent only - // when the role set a time cap; otherwise the context deadline (= - // query_timeout) is the time bound the driver derives. - limits := chQueryLimits{ - MaxResultRows: perms.Select.MaxRows, - MaxRowsToRead: perms.Select.MaxRowsToRead, - MaxMemoryBytes: perms.Select.MaxMemoryUsage.Bytes(), - } - if timeCap > 0 { - limits.ExecutionTime = timeout - } - if settings := chReadSettings(limits); settings != nil { - queryCtx = clickhouse.Context(queryCtx, clickhouse.WithSettings(settings)) - } - start := time.Now() - - rows, err := executeCHQuery(queryCtx, conn, result.SQL, result.Params) - queryDuration := time.Since(start) + // ClickHouse's own JSON rendering of the rows, stored and served + // verbatim — no per-row scan, no re-marshal. + data, err := h.ch.do(queryCtx, target, conns, chRequest{sql: chSQL, params: chParams, settings: chSettings}) if err != nil { - // TODO: depending on the error, we may actually want to cache it return nil, err } - - data, err := json.Marshal(rows) - if err != nil { - // TODO: eventually we want CSV support etc - return nil, err - } - - ttl := cache.QueryTimeToTTL(queryDuration) - if h.Cache != nil { - _ = h.Cache.Set(r.Context(), snap, data, ttl) + _ = h.Cache.Set(r.Context(), snap, data, cache.QueryTimeToTTL(time.Since(start))) } return data, nil }) if err != nil { - writeCHError(w, r, err, err.Error(), http.StatusInternalServerError, caps) + writeCHError(w, r, err, chErrorMessage(err), http.StatusInternalServerError, caps) return } w.Header().Set("Content-Type", "application/json") w.Header().Set("X-Cache", "MISS") - _, _ = w.Write(v.([]byte)) //nolint:gosec // G705: the tenant id on the key only selects the entry; the bytes are JSON the handler marshalled from ClickHouse rows + _, _ = w.Write(v.([]byte)) //nolint:gosec // G705: the tenant id on the key only selects the entry; the bytes are ClickHouse's JSONEachRow rows framed as an array } diff --git a/internal/api/structured_query_test.go b/internal/api/structured_query_test.go index c715502b..f3c3774c 100644 --- a/internal/api/structured_query_test.go +++ b/internal/api/structured_query_test.go @@ -4,14 +4,16 @@ import ( "bytes" "context" "encoding/json" + "fmt" "net/http" "net/http/httptest" "net/url" + "strings" "testing" "time" - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/Wave-RF/WaveHouse/internal/auth" + "github.com/Wave-RF/WaveHouse/internal/cache" "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/query" @@ -259,26 +261,12 @@ func TestStructuredQuery_NilPolicyFailsClosed(t *testing.T) { // ─── #223: column allowlist is a hard cap on every read, end-to-end ────────── -// sqlCapturingConn records the SQL (and bound args) the handler hands to -// ClickHouse so tests can assert the generated query without a live database. -// Query returns an empty result set (the handler marshals it to []); these -// tests assert on the SQL string, args, and HTTP status, not on rows. lastSQL -// stays empty when the request is rejected before execution — which is itself -// the assertion for denied paths. -type sqlCapturingConn struct { - driver.Conn - lastSQL string - lastArgs []any -} - -func (c *sqlCapturingConn) Query(_ context.Context, sql string, args ...any) (driver.Rows, error) { - c.lastSQL = sql - c.lastArgs = args - return &chainEmptyRows{}, nil -} - -// sensitiveSchema has a column (payload, user_id) that restrictive policies hide. -func newCapturingHandler(t *testing.T, conn driver.Conn, p *policy.Policy) *StructuredQueryHandler { +// newCapturingHandler is a handler over a schema with columns (payload, +// user_id) that restrictive policies hide, reading through ch — which records +// the SQL and bound values it is sent and answers with no rows unless told +// otherwise. ch.sql() stays empty when a request is rejected before +// execution, which is itself the assertion for the denied paths. +func newCapturingHandler(t *testing.T, ch *fakeCH, p *policy.Policy) *StructuredQueryHandler { t.Helper() reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ { @@ -291,7 +279,9 @@ func newCapturingHandler(t *testing.T, conn driver.Conn, p *policy.Policy) *Stru }, }, }) - return NewStructuredQueryHandler(fixedConn(conn), nil, fixedRegistry(reg), staticPolicy(p), func(*settings.Store) int { return 60 }, func(*settings.Store) time.Duration { return 5 * time.Second }, nil) + h := NewStructuredQueryHandler(ch.target, nil, fixedRegistry(reg), staticPolicy(p), func(*settings.Store) int { return 60 }, func(*settings.Store) time.Duration { return 5 * time.Second }, nil) + h.ch = ch.reader() + return h } func viewerRequest(t *testing.T, sq query.StructuredQuery) *http.Request { @@ -315,17 +305,17 @@ func policyWithViewer(perms policy.SelectPermissions) *policy.Policy { // payload/user_id never reach ClickHouse — let alone the client. func TestStructuredQuery_SelectAll_RestrictedRoleGetsAllowedProjection(t *testing.T) { t.Parallel() - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"page", "ts"}})) + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"page", "ts"}})) w := httptest.NewRecorder() h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{SelectAll: true}))) require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) - assert.Equal(t, "SELECT `page`, `ts` FROM `clicks` LIMIT 10000", conn.lastSQL) - assert.NotContains(t, conn.lastSQL, "*") - assert.NotContains(t, conn.lastSQL, "payload") - assert.NotContains(t, conn.lastSQL, "user_id") + assert.Equal(t, "SELECT `page`, `ts` FROM `clicks` LIMIT 10000", ch.sql()) + assert.NotContains(t, ch.sql(), "*") + assert.NotContains(t, ch.sql(), "payload") + assert.NotContains(t, ch.sql(), "user_id") } // TestStructuredQuery_RowFilterAndMaxRows_ReachClickHouse pins the handler seam @@ -338,8 +328,8 @@ func TestStructuredQuery_SelectAll_RestrictedRoleGetsAllowedProjection(t *testin func TestStructuredQuery_RowFilterAndMaxRows_ReachClickHouse(t *testing.T) { t.Parallel() eq := "{{ jwt.org_id }}" - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, policyWithViewer(policy.SelectPermissions{ + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{ Filter: map[string]policy.Filter{"user_id": {Eq: &eq}}, MaxRows: 100, })) @@ -354,8 +344,8 @@ func TestStructuredQuery_RowFilterAndMaxRows_ReachClickHouse(t *testing.T) { h.Handle(w, withTenant(r.WithContext(ctx))) require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) - assert.Equal(t, "SELECT `page` FROM `clicks` WHERE (`user_id` = ?) AND `page` = ? LIMIT 100", conn.lastSQL) - assert.Equal(t, []any{"org-1", "/home"}, conn.lastArgs) + assert.Equal(t, "SELECT `page` FROM `clicks` WHERE (`user_id` = {p0:String}) AND `page` = {p1:String} LIMIT 100", ch.sql()) + assert.Equal(t, []string{"org-1", "/home"}, ch.params()) } // TestStructuredQuery_OmittedColumns_ReturnsNothing pins safe-by-default: a request @@ -364,30 +354,30 @@ func TestStructuredQuery_RowFilterAndMaxRows_ReachClickHouse(t *testing.T) { // simply leaving columns out. func TestStructuredQuery_OmittedColumns_ReturnsNothing(t *testing.T) { t.Parallel() - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"page", "ts"}})) + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"page", "ts"}})) w := httptest.NewRecorder() h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{}))) require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) assert.JSONEq(t, "[]", w.Body.String()) - assert.Empty(t, conn.lastSQL, "an empty projection must not reach ClickHouse") + assert.Empty(t, ch.sql(), "an empty projection must not reach ClickHouse") } // TestStructuredQuery_SelectAll_DenyListExpands: select_all under a deny-list // (empty allow) expands to the non-denied columns, never a raw SELECT *. func TestStructuredQuery_SelectAll_DenyListExpands(t *testing.T) { t.Parallel() - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, policyWithViewer(policy.SelectPermissions{DenyColumns: []string{"payload"}})) + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{DenyColumns: []string{"payload"}})) w := httptest.NewRecorder() h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{SelectAll: true}))) require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) - assert.Equal(t, "SELECT `page`, `user_id`, `ts` FROM `clicks` LIMIT 10000", conn.lastSQL) - assert.NotContains(t, conn.lastSQL, "payload") + assert.Equal(t, "SELECT `page`, `user_id`, `ts` FROM `clicks` LIMIT 10000", ch.sql()) + assert.NotContains(t, ch.sql(), "payload") } // TestStructuredQuery_LiteralStarColumn_Unknown: columns:["*"] is a literal column @@ -395,14 +385,14 @@ func TestStructuredQuery_SelectAll_DenyListExpands(t *testing.T) { // column — the all-columns wildcard is select_all. func TestStructuredQuery_LiteralStarColumn_Unknown(t *testing.T) { t.Parallel() - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) w := httptest.NewRecorder() h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{Columns: []string{"*"}}))) require.Equal(t, http.StatusBadRequest, w.Code, "body=%s", w.Body.String()) - assert.Empty(t, conn.lastSQL) + assert.Empty(t, ch.sql()) } // TestStructuredQuery_UnrestrictedRoleKeepsSelectStar proves the common case is @@ -410,14 +400,14 @@ func TestStructuredQuery_LiteralStarColumn_Unknown(t *testing.T) { // columns and admin convenience preserved; no behavior change off the hot path). func TestStructuredQuery_UnrestrictedRoleKeepsSelectStar(t *testing.T) { t.Parallel() - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) w := httptest.NewRecorder() h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{SelectAll: true}))) require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) - assert.Equal(t, "SELECT * FROM `clicks` LIMIT 10000", conn.lastSQL) + assert.Equal(t, "SELECT * FROM `clicks` LIMIT 10000", ch.sql()) } // TestStructuredQuery_DeniedColumnInAnyClause_Returns403 is the regression for @@ -456,15 +446,15 @@ func TestStructuredQuery_DeniedColumnInAnyClause_Returns403(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { t.Parallel() - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"page", "ts"}})) + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"page", "ts"}})) w := httptest.NewRecorder() h.Handle(w, withTenant(viewerRequest(t, tt.sq))) assert.Equal(t, http.StatusForbidden, w.Code, "body=%s", w.Body.String()) assert.Contains(t, w.Body.String(), "not allowed") - assert.Empty(t, conn.lastSQL, "a denied query must never reach ClickHouse") + assert.Empty(t, ch.sql(), "a denied query must never reach ClickHouse") testutil.AssertJSONErrorResponse(t, w) }) } @@ -475,14 +465,14 @@ func TestStructuredQuery_DeniedColumnInAnyClause_Returns403(t *testing.T) { // a fail-open SELECT *. func TestStructuredQuery_NoReadableColumns_Returns403(t *testing.T) { t.Parallel() - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"nonexistent"}})) + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"nonexistent"}})) w := httptest.NewRecorder() h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{SelectAll: true}))) assert.Equal(t, http.StatusForbidden, w.Code, "body=%s", w.Body.String()) - assert.Empty(t, conn.lastSQL) + assert.Empty(t, ch.sql()) testutil.AssertJSONErrorResponse(t, w) } @@ -491,10 +481,10 @@ func TestStructuredQuery_NoReadableColumns_Returns403(t *testing.T) { // and must get only that role's columns, not every column. func TestStructuredQuery_UnauthenticatedUsesDefaultRoleProjection(t *testing.T) { t.Parallel() - conn := &sqlCapturingConn{} + ch := &fakeCH{} p := policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"page"}}) p.DefaultRole = "viewer" // public access resolves to the restricted viewer role - h := newCapturingHandler(t, conn, p) + h := newCapturingHandler(t, ch, p) // No role on the context — a tokenless request. r := structuredQueryRequest(t, "clicks", query.StructuredQuery{SelectAll: true, Limit: 2}) @@ -502,8 +492,8 @@ func TestStructuredQuery_UnauthenticatedUsesDefaultRoleProjection(t *testing.T) h.Handle(w, withTenant(r)) require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) - assert.Equal(t, "SELECT `page` FROM `clicks` LIMIT 2", conn.lastSQL) - assert.NotContains(t, conn.lastSQL, "payload") + assert.Equal(t, "SELECT `page` FROM `clicks` LIMIT 2", ch.sql()) + assert.NotContains(t, ch.sql(), "payload") } // TestStructuredQuery_CacheKeyIsolatesColumnVisibility pins the cache-isolation @@ -521,17 +511,149 @@ func TestStructuredQuery_CacheKeyIsolatesColumnVisibility(t *testing.T) { }, }} sqlFor := func(role string) string { - conn := &sqlCapturingConn{} - h := newCapturingHandler(t, conn, p) + ch := &fakeCH{} + h := newCapturingHandler(t, ch, p) r := structuredQueryRequest(t, "clicks", query.StructuredQuery{SelectAll: true}) r = r.WithContext(auth.WithClaims(auth.WithRole(r.Context(), role), jwt.MapClaims{})) w := httptest.NewRecorder() h.Handle(w, withTenant(r)) require.Equal(t, http.StatusOK, w.Code, "role=%s body=%s", role, w.Body.String()) - return conn.lastSQL + return ch.sql() } viewerSQL, auditorSQL := sqlFor("viewer"), sqlFor("auditor") assert.NotEqual(t, viewerSQL, auditorSQL) assert.NotEqual(t, queryCacheKey(tenant.Default, viewerSQL, nil), queryCacheKey(tenant.Default, auditorSQL, nil), "roles with different column visibility must not share a cache key") } + +// TestStructuredQuery_ResourceCapsReachTheWire pins that the role's caps are +// sent as ClickHouse settings on the read's URL (#316). The integration suite +// proves ClickHouse honours them; this proves they are sent at all, which is +// the half that silently regresses. +func TestStructuredQuery_ResourceCapsReachTheWire(t *testing.T) { + t.Parallel() + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{ + AllowColumns: []string{"page"}, + MaxRows: 50, + MaxRowsToRead: 1234, + MaxMemoryUsage: 1 << 20, + })) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{SelectAll: true}))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, "50", ch.setting("max_result_rows")) + assert.Equal(t, "throw", ch.setting("result_overflow_mode")) + assert.Equal(t, "1234", ch.setting("max_rows_to_read")) + assert.Equal(t, "throw", ch.setting("read_overflow_mode")) + assert.Equal(t, "1048576", ch.setting("max_memory_usage")) + assert.Equal(t, "2", ch.setting("readonly"), "a structured query is a read") +} + +// TestStructuredQuery_TimeBoundReachesClickHouse: every read carries +// max_execution_time — the role's time cap when it is the tighter budget, +// query_timeout otherwise — so ClickHouse stops a query nobody waits for and +// says which limit stopped it. The handler's query_timeout is 5s. +func TestStructuredQuery_TimeBoundReachesClickHouse(t *testing.T) { + t.Parallel() + for _, tc := range []struct { + name string + capMilli policy.Millis + want string + }{ + {"no cap is query_timeout", 0, "5"}, + {"a tighter cap", 500, "0.5"}, + {"a looser cap is query_timeout", 10000, "5"}, + } { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}, MaxExecutionTime: tc.capMilli})) + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{Columns: []string{"page"}}))) + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + assert.Equal(t, tc.want, ch.setting("max_execution_time")) + }) + } +} + +// TestStructuredQuery_ServesClickHouseBytes pins the response path end to end: +// ClickHouse's own JSON rendering is what the caller gets and what the cache +// stores, with no re-marshal in between, and the second read is served from +// the cache byte-for-byte under X-Cache: HIT. +func TestStructuredQuery_ServesClickHouseBytes(t *testing.T) { + t.Parallel() + // A body only ClickHouse would produce: keys in SELECT order, a Decimal + // as a bare number with its digits, and a DateTime in ClickHouse's own + // spelling. A round trip through map[string]any would reorder the keys. + const row = `{"page":"/home","amount":12.50,"ts":"2026-01-15 10:30:00"}` + ch := &fakeCH{answer: answerRows(row + "\n")} + c, err := cache.NewLocal(1 << 20) + require.NoError(t, err) + t.Cleanup(func() { _ = c.Close() }) + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) + h.Cache = c + + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{SelectAll: true}))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, "MISS", w.Header().Get("X-Cache")) + assert.Equal(t, "["+row+"]", w.Body.String()) + + c.Wait() + hit := httptest.NewRecorder() + h.Handle(hit, withTenant(viewerRequest(t, query.StructuredQuery{SelectAll: true}))) + assert.Equal(t, "HIT", hit.Header().Get("X-Cache")) + assert.Equal(t, w.Body.String(), hit.Body.String(), "a cache hit must be byte-identical to the miss") + assert.Equal(t, int32(1), ch.reads.Load()) +} + +// TestStructuredQuery_FilterValuesKeepTheirDigits: a filter value reaches +// ClickHouse as the text the caller wrote — a decimal's trailing zero, an +// integer past 2^53 — not as a float64's rendering of it. +func TestStructuredQuery_FilterValuesKeepTheirDigits(t *testing.T) { + t.Parallel() + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) + r := httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/v1/query?table=clicks", + strings.NewReader(`{"columns":["page"],"filters":[{"column":"payload","op":"eq","value":12.50},{"column":"user_id","op":"in","value":[9007199254740993]}]}`)) + r = r.WithContext(auth.WithClaims(auth.WithRole(r.Context(), "viewer"), jwt.MapClaims{})) + w := httptest.NewRecorder() + h.Handle(w, withTenant(r)) + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + assert.Equal(t, []string{"12.50", "['9007199254740993']"}, ch.params()) +} + +// TestStructuredQuery_UnbindableFilterValueIs400: a filter value with no +// honest binding — a JSON null, or an `in` list too large for ClickHouse's +// HTTP interface to take — is the caller's malformed query, a 400 before +// anything reaches ClickHouse, rather than a server error after. +func TestStructuredQuery_UnbindableFilterValueIs400(t *testing.T) { + t.Parallel() + huge := make([]any, 0, 20000) + for i := range 20000 { + huge = append(huge, fmt.Sprintf("v%d", i)) + } + for _, tc := range []struct { + name string + filter query.Filter + want string + }{ + {"null value", query.Filter{Column: "page", Op: "eq", Value: nil}, "must not be null"}, + {"oversized in list", query.Filter{Column: "page", Op: "in", Value: huge}, "split a long in list"}, + } { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{Columns: []string{"page"}, Filters: []query.Filter{tc.filter}}))) + require.Equal(t, http.StatusBadRequest, w.Code, w.Body.String()) + assert.Contains(t, w.Body.String(), tc.want) + testutil.AssertJSONErrorResponse(t, w) + assert.Empty(t, ch.sql(), "an unbindable query must never reach ClickHouse") + }) + } +} diff --git a/internal/api/tenant_clickhouse_test.go b/internal/api/tenant_clickhouse_test.go index 781b3e4a..bc562a9a 100644 --- a/internal/api/tenant_clickhouse_test.go +++ b/internal/api/tenant_clickhouse_test.go @@ -119,18 +119,18 @@ func TestClickHouseRoutes_NoPoolIs503(t *testing.T) { DefaultRole: "viewer", Tables: map[string]policy.TablePolicy{"clicks": {"viewer": {Select: &policy.SelectPermissions{AllowColumns: []string{"page"}}}}}, }) - noConn := func(*settings.Store) driver.Conn { return nil } + noTarget := func(*settings.Store) chconn.Target { return chconn.Target{} } t.Run("structured query", func(t *testing.T) { t.Parallel() - h := NewStructuredQueryHandler(noConn, nil, fixedRegistry(reg), allowAll, nil, noTimeout, nil) + h := NewStructuredQueryHandler(noTarget, nil, fixedRegistry(reg), allowAll, nil, noTimeout, nil) w := httptest.NewRecorder() h.Handle(w, withTenant(structuredQueryRequest(t, "clicks", selectAllQuery()))) assertUnavailable(t, w, noConnectionMessage, retryAfterPool) }) t.Run("pipe execute", func(t *testing.T) { t.Parallel() - h := NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), allowAll, noConn, nil, noTimeout) + h := NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), allowAll, noTarget, nil, noTimeout) w := httptest.NewRecorder() h.Execute(w, withTenant(pipesRequest(t, http.MethodGet, "/v1/pipes/top_pages", "top_pages", nil))) assertUnavailable(t, w, noConnectionMessage, retryAfterPool) @@ -139,7 +139,7 @@ func TestClickHouseRoutes_NoPoolIs503(t *testing.T) { // keeps its Retry-After where a failed write's answer drops it. t.Run("write pipe execute", func(t *testing.T) { t.Parallel() - h := NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "log", SQL: "INSERT INTO audit_log VALUES (1)", AllowedRoles: []string{"viewer"}}), allowAll, noConn, nil, noTimeout) + h := NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "log", SQL: "INSERT INTO audit_log VALUES (1)", AllowedRoles: []string{"viewer"}}), allowAll, noTarget, nil, noTimeout) w := httptest.NewRecorder() h.Execute(w, withTenant(pipesRequest(t, http.MethodGet, "/v1/pipes/log", "log", nil))) assertUnavailable(t, w, noConnectionMessage, retryAfterPool) @@ -249,8 +249,8 @@ func TestClickHouseOpsRoutes_TenantParam(t *testing.T) { } } -// The ClickHouse-side getters — the connection, the registry, the HTTP -// target and the query deadline — receive the request's own tenant store +// The ClickHouse-side getters — the registry, the HTTP target and the query +// deadline — receive the request's own tenant store // through the real router, on the routes that reach ClickHouse: two tenants // alternating never hand one the other's. func TestNewRouter_ClickHouseGettersReceiveTheRequestTenantsStore(t *testing.T) { @@ -262,14 +262,19 @@ func TestNewRouter_ClickHouseGettersReceiveTheRequestTenantsStore(t *testing.T) DefaultRole: "viewer", Tables: map[string]policy.TablePolicy{"clicks": {"viewer": {Select: &policy.SelectPermissions{AllowColumns: []string{"page"}}}}}, }) - conn := func(s *settings.Store) driver.Conn { record(s); return &countingConn{} } + ch := &fakeCH{} + target := func(s *settings.Store) chconn.Target { record(s); return ch.target(s) } registry := func(s *settings.Store) *discovery.SchemaRegistry { record(s); return reg } timeout := func(s *settings.Store) time.Duration { record(s); return time.Second } + sq := NewStructuredQueryHandler(target, nil, registry, viewer, func(*settings.Store) int { return 60 }, timeout, nil) + sq.ch = ch.reader() + pipesHandler := NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), viewer, target, nil, timeout) + pipesHandler.ch = ch.reader() router := NewRouter(Dependencies{ Tenants: tenants, Ingest: NewIngestHandler(registry, &testutil.MockPublisher{}), - StructuredQuery: NewStructuredQueryHandler(conn, nil, registry, viewer, func(*settings.Store) int { return 60 }, timeout, nil), - Pipes: NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), viewer, conn, nil, timeout), + StructuredQuery: sq, + Pipes: pipesHandler, Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, diff --git a/internal/api/tenant_helpers_test.go b/internal/api/tenant_helpers_test.go index d48cd2e5..7c0c02ec 100644 --- a/internal/api/tenant_helpers_test.go +++ b/internal/api/tenant_helpers_test.go @@ -7,7 +7,6 @@ import ( "path/filepath" "testing" - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/stretchr/testify/require" "github.com/Wave-RF/WaveHouse/internal/dedupe" @@ -65,11 +64,6 @@ func schemaHandlerOver(reg *discovery.SchemaRegistry, tenants *settings.Registry return h } -// fixedConn is a connection source fixed to conn, whatever the tenant. -func fixedConn(conn driver.Conn) func(*settings.Store) driver.Conn { - return func(*settings.Store) driver.Conn { return conn } -} - // staticPolicy is a PolicySource fixed to p, whatever the tenant. func staticPolicy(p *policy.Policy) PolicySource { return func(*settings.Store) *policy.Policy { return p } diff --git a/internal/app/wire.go b/internal/app/wire.go index 70ada1bb..e59660cf 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -289,7 +289,8 @@ func (a *App) wireObservability(ctx context.Context) { // had, or on none when they had none; both logged, and retried by the next // reload. Reachability surfaces // where it already does (schema discovery retries, /readyz, query errors). -// Every consumer resolves its tenant's pool per call (chConn, chTargetFor). +// Every consumer resolves its tenant's pool per call (chTargetFor, +// discoverySource). func (a *App) wireClickHouse() error { members := func() []chconn.Member { var ms []chconn.Member @@ -340,17 +341,6 @@ func (a *App) wireClickHouse() error { return nil } -// chConn is the connection of tenant id, or an untyped nil when the tenant -// is on no pool — never a nil *Manager inside a non-nil driver.Conn, which -// would pass a nil check and panic on use. -func (a *App) chConn(id tenant.ID) driver.Conn { - m := a.pools.For(id) - if m == nil { - return nil - } - return m -} - // discoverySource is what tenant id's schema registry discovers from, read // per refresh so a reload that repoints the tenant applies to the next one // (a move to another address or database starts the tenant over on a fresh @@ -372,8 +362,6 @@ func (a *App) discoverySource(id tenant.ID) discovery.Source { // tenant's pool or registry per call, so a reload that repoints the tenant // applies to the next request. -func (a *App) chConnFor(s *settings.Store) driver.Conn { return a.chConn(s.Tenant()) } - func (a *App) chTargetFor(s *settings.Store) chconn.Target { return a.pools.Target(s.Tenant()) } func (a *App) registryFor(s *settings.Store) *discovery.SchemaRegistry { @@ -1011,7 +999,7 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { closing := make(chan struct{}) streamHandler.Closing = closing - pipesHandler := api.NewPipesHandler(func(s *settings.Store) pipes.Source { return s }, (*settings.Store).Policy, a.chConnFor, a.cache, queryTimeout) + pipesHandler := api.NewPipesHandler(func(s *settings.Store) pipes.Source { return s }, (*settings.Store).Policy, a.chTargetFor, a.cache, queryTimeout) pipesHandler.Tenants = a.tenants schemaHandler := api.NewSchemaHandler(a.registryFor) @@ -1032,7 +1020,7 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { Schema: schemaHandler, DLQ: api.NewDLQHandler(a.mq), Pipes: pipesHandler, - StructuredQuery: api.NewStructuredQueryHandler(a.chConnFor, a.cache, a.registryFor, (*settings.Store).Policy, (*settings.Store).TimestampBucketSeconds, queryTimeout, (*settings.Store).DefaultMaxRows), + StructuredQuery: api.NewStructuredQueryHandler(a.chTargetFor, a.cache, a.registryFor, (*settings.Store).Policy, (*settings.Store).TimestampBucketSeconds, queryTimeout, (*settings.Store).DefaultMaxRows), AuthMW: authMW, Tenants: a.tenants, diff --git a/internal/pipes/pipes.go b/internal/pipes/pipes.go index 28e8b44d..8e7fb480 100644 --- a/internal/pipes/pipes.go +++ b/internal/pipes/pipes.go @@ -82,9 +82,11 @@ func isNumericLiteral(s string) bool { // parameter definitions. // // Values are inlined directly into the SQL string (scalars are escaped, arrays -// render as a parenthesized list — see formatParamValue). This avoids -// driver-level positional parameter limitations (e.g. LIMIT position). -func BindParams(q *NamedQuery, supplied map[string]any) (string, []any, error) { +// render as a parenthesized list — see formatParamValue). This avoids the +// positional-parameter limitations a pipe would otherwise hit: a placeholder +// may sit in a LIMIT, a FORMAT clause or an identifier, where a bound value is +// not legal SQL. Nothing is left for the caller to bind. +func BindParams(q *NamedQuery, supplied map[string]any) (string, error) { // Build lookup from formal parameter definitions. formal := make(map[string]*ParamDef, len(q.Parameters)) for i := range q.Parameters { @@ -95,7 +97,7 @@ func BindParams(q *NamedQuery, supplied map[string]any) (string, []any, error) { for _, p := range q.Parameters { if p.Required { if _, ok := supplied[p.Name]; !ok { - return "", nil, fmt.Errorf("missing required parameter: %s", p.Name) + return "", fmt.Errorf("missing required parameter: %s", p.Name) } } } @@ -132,9 +134,9 @@ func BindParams(q *NamedQuery, supplied map[string]any) (string, []any, error) { }) if bindErr != nil { - return "", nil, bindErr + return "", bindErr } - return sql, nil, nil + return sql, nil } // formatParamValue converts a Go value to a safe SQL literal for inline diff --git a/internal/pipes/pipes_test.go b/internal/pipes/pipes_test.go index d5314d9e..d35e2335 100644 --- a/internal/pipes/pipes_test.go +++ b/internal/pipes/pipes_test.go @@ -16,10 +16,9 @@ func TestBindParams_AllSupplied(t *testing.T) { {Name: "min_count", Type: "number", Required: true}, }, } - sql, params, err := BindParams(q, map[string]any{"page": "/home", "min_count": 10}) + sql, err := BindParams(q, map[string]any{"page": "/home", "min_count": 10}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM clicks WHERE page = '/home' AND count > 10", sql) - assert.Nil(t, params) } func TestBindParams_MissingRequired(t *testing.T) { @@ -28,7 +27,7 @@ func TestBindParams_MissingRequired(t *testing.T) { SQL: "SELECT * FROM clicks WHERE page = {{page}}", Parameters: []ParamDef{{Name: "page", Type: "string", Required: true}}, } - _, _, err := BindParams(q, map[string]any{}) + _, err := BindParams(q, map[string]any{}) assert.Error(t, err) assert.Contains(t, err.Error(), "missing required parameter: page") } @@ -39,10 +38,9 @@ func TestBindParams_DefaultApplied(t *testing.T) { SQL: "SELECT * FROM clicks LIMIT {{limit}}", Parameters: []ParamDef{{Name: "limit", Type: "number", Default: 100}}, } - sql, params, err := BindParams(q, map[string]any{}) + sql, err := BindParams(q, map[string]any{}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM clicks LIMIT 100", sql) - assert.Nil(t, params) } func TestBindParams_MultipleOccurrences(t *testing.T) { @@ -51,19 +49,17 @@ func TestBindParams_MultipleOccurrences(t *testing.T) { SQL: "SELECT * FROM t WHERE a = {{val}} OR b = {{val}}", Parameters: []ParamDef{{Name: "val", Type: "string", Required: true}}, } - sql, params, err := BindParams(q, map[string]any{"val": "x"}) + sql, err := BindParams(q, map[string]any{"val": "x"}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM t WHERE a = 'x' OR b = 'x'", sql) - assert.Nil(t, params) } func TestBindParams_NoParameters(t *testing.T) { t.Parallel() q := &NamedQuery{SQL: "SELECT count(*) FROM clicks"} - sql, params, err := BindParams(q, map[string]any{}) + sql, err := BindParams(q, map[string]any{}) require.NoError(t, err) assert.Equal(t, "SELECT count(*) FROM clicks", sql) - assert.Empty(t, params) } func TestBindParams_OptionalWithDefault_Supplied(t *testing.T) { @@ -72,10 +68,9 @@ func TestBindParams_OptionalWithDefault_Supplied(t *testing.T) { SQL: "SELECT * FROM clicks LIMIT {{limit}}", Parameters: []ParamDef{{Name: "limit", Type: "number", Default: 100}}, } - sql, params, err := BindParams(q, map[string]any{"limit": 50}) + sql, err := BindParams(q, map[string]any{"limit": 50}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM clicks LIMIT 50", sql) - assert.Nil(t, params) } func TestBindParams_PlaceholderNotInSQL(t *testing.T) { @@ -84,10 +79,9 @@ func TestBindParams_PlaceholderNotInSQL(t *testing.T) { SQL: "SELECT * FROM clicks", Parameters: []ParamDef{{Name: "unused", Type: "string"}}, } - sql, params, err := BindParams(q, map[string]any{"unused": "val"}) + sql, err := BindParams(q, map[string]any{"unused": "val"}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM clicks", sql) - assert.Empty(t, params, "unused param should not generate positional args") } func TestBindParams_InlineDefault_NoFormalParam(t *testing.T) { @@ -95,10 +89,9 @@ func TestBindParams_InlineDefault_NoFormalParam(t *testing.T) { q := &NamedQuery{ SQL: "SELECT page, count() FROM clicks GROUP BY page LIMIT {{limit:10}}", } - sql, params, err := BindParams(q, map[string]any{}) + sql, err := BindParams(q, map[string]any{}) require.NoError(t, err) assert.Equal(t, "SELECT page, count() FROM clicks GROUP BY page LIMIT 10", sql) - assert.Nil(t, params) } func TestBindParams_InlineDefault_SuppliedOverrides(t *testing.T) { @@ -106,10 +99,9 @@ func TestBindParams_InlineDefault_SuppliedOverrides(t *testing.T) { q := &NamedQuery{ SQL: "SELECT * FROM clicks LIMIT {{limit:10}}", } - sql, params, err := BindParams(q, map[string]any{"limit": float64(5)}) + sql, err := BindParams(q, map[string]any{"limit": float64(5)}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM clicks LIMIT 5", sql) - assert.Nil(t, params) } func TestBindParams_InlineNoDefault_MissingRequired(t *testing.T) { @@ -117,7 +109,7 @@ func TestBindParams_InlineNoDefault_MissingRequired(t *testing.T) { q := &NamedQuery{ SQL: "SELECT * FROM clicks WHERE page = {{page}}", } - _, _, err := BindParams(q, map[string]any{}) + _, err := BindParams(q, map[string]any{}) assert.Error(t, err) assert.Contains(t, err.Error(), "missing required parameter: page") } @@ -127,10 +119,9 @@ func TestBindParams_InlineMultipleParams(t *testing.T) { q := &NamedQuery{ SQL: "SELECT * FROM clicks WHERE country = {{country:US}} LIMIT {{limit:10}}", } - sql, params, err := BindParams(q, map[string]any{}) + sql, err := BindParams(q, map[string]any{}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM clicks WHERE country = 'US' LIMIT 10", sql) - assert.Nil(t, params) } func TestBindParams_StringEscaping(t *testing.T) { @@ -139,7 +130,7 @@ func TestBindParams_StringEscaping(t *testing.T) { SQL: "SELECT * FROM t WHERE name = {{name}}", Parameters: []ParamDef{{Name: "name", Type: "string", Required: true}}, } - sql, _, err := BindParams(q, map[string]any{"name": "O'Brien"}) + sql, err := BindParams(q, map[string]any{"name": "O'Brien"}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM t WHERE name = 'O''Brien'", sql) } @@ -150,7 +141,7 @@ func TestBindParams_BooleanParam(t *testing.T) { SQL: "SELECT * FROM t WHERE active = {{active}}", Parameters: []ParamDef{{Name: "active", Type: "boolean", Required: true}}, } - sql, _, err := BindParams(q, map[string]any{"active": true}) + sql, err := BindParams(q, map[string]any{"active": true}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM t WHERE active = 1", sql) } @@ -161,7 +152,7 @@ func TestBindParams_NilParam(t *testing.T) { SQL: "SELECT * FROM t WHERE col = {{val}}", Parameters: []ParamDef{{Name: "val", Type: "string", Default: nil}}, } - sql, _, err := BindParams(q, map[string]any{}) + sql, err := BindParams(q, map[string]any{}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM t WHERE col = NULL", sql) } @@ -231,17 +222,16 @@ func TestBindParams_ArrayInClause(t *testing.T) { SQL: "SELECT * FROM t WHERE id IN {{ids}}", Parameters: []ParamDef{{Name: "ids", Type: "array", Required: true}}, } - sql, params, err := BindParams(q, map[string]any{"ids": []any{"a", "b", "c"}}) + sql, err := BindParams(q, map[string]any{"ids": []any{"a", "b", "c"}}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM t WHERE id IN ('a', 'b', 'c')", sql) - assert.Nil(t, params) } func TestBindParams_ArrayWorksWithoutDeclaredType(t *testing.T) { t.Parallel() // An undeclared (inline) parameter still renders an array safely. q := &NamedQuery{SQL: "SELECT * FROM t WHERE id IN {{ids}}"} - sql, _, err := BindParams(q, map[string]any{"ids": []any{float64(1), float64(2)}}) + sql, err := BindParams(q, map[string]any{"ids": []any{float64(1), float64(2)}}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM t WHERE id IN (1, 2)", sql) } @@ -252,7 +242,7 @@ func TestBindParams_ArrayMixedNumericAndStringElements(t *testing.T) { // quoted on its own: a numeric-looking string renders bare, a non-numeric one // is quoted (and escaped). q := &NamedQuery{SQL: "SELECT * FROM t WHERE id IN {{ids}}"} - sql, _, err := BindParams(q, map[string]any{"ids": []any{"100", "abc"}}) + sql, err := BindParams(q, map[string]any{"ids": []any{"100", "abc"}}) require.NoError(t, err) assert.Equal(t, "SELECT * FROM t WHERE id IN (100, 'abc')", sql) } @@ -263,7 +253,7 @@ func TestBindParams_ArrayMixedNumericAndStringElements(t *testing.T) { func TestBindParams_ArrayNeutralizesInjection(t *testing.T) { t.Parallel() q := &NamedQuery{SQL: "SELECT secret FROM t WHERE id IN {{ids}}"} - sql, _, err := BindParams(q, map[string]any{ + sql, err := BindParams(q, map[string]any{ "ids": []any{"' UNION SELECT pw FROM users -- "}, }) require.NoError(t, err) @@ -273,7 +263,7 @@ func TestBindParams_ArrayNeutralizesInjection(t *testing.T) { func TestBindParams_ObjectRejected(t *testing.T) { t.Parallel() q := &NamedQuery{SQL: "SELECT * FROM t WHERE col = {{p}}"} - _, _, err := BindParams(q, map[string]any{"p": map[string]any{"k": "v"}}) + _, err := BindParams(q, map[string]any{"p": map[string]any{"k": "v"}}) require.Error(t, err) assert.Contains(t, err.Error(), `parameter "p"`) assert.Contains(t, err.Error(), "unsupported parameter type object") @@ -282,7 +272,7 @@ func TestBindParams_ObjectRejected(t *testing.T) { func TestBindParams_EmptyArrayRejected(t *testing.T) { t.Parallel() q := &NamedQuery{SQL: "SELECT * FROM t WHERE id IN {{ids}}"} - _, _, err := BindParams(q, map[string]any{"ids": []any{}}) + _, err := BindParams(q, map[string]any{"ids": []any{}}) require.Error(t, err) assert.Contains(t, err.Error(), "array parameter must not be empty") } diff --git a/internal/query/builder.go b/internal/query/builder.go index 9851e655..49aa6e85 100644 --- a/internal/query/builder.go +++ b/internal/query/builder.go @@ -1,6 +1,7 @@ package query import ( + "encoding/json" "fmt" "regexp" "strconv" @@ -20,7 +21,10 @@ import ( // misconfigured to 0. const DefaultMaxRows = 10000 -// BuildResult holds the generated SQL and bound parameters. +// BuildResult holds the generated SQL and its bound values. The SQL carries +// positional `?` placeholders, one per entry in Params, in left-to-right +// order. NamedParams turns that pair into the named-parameter form +// ClickHouse's HTTP interface takes. type BuildResult struct { SQL string Params []any @@ -43,7 +47,7 @@ type BuildResult struct { // Every identifier that reaches the SQL — columns, the table, aggregation // aliases — is backtick-quoted via chsql.QuoteIdent, so the builder accepts any // name ClickHouse accepts while remaining injection-safe. Values stay positional -// `?` parameters bound by the driver. +// `?` parameters, which NamedParams turns into ClickHouse named parameters. // // Projection rules: SelectAll requests every readable column (expanded to the // role's allow/deny set); an explicit Columns list projects exactly those (where @@ -101,7 +105,7 @@ func Build(table string, q *StructuredQuery, schema *discovery.TableSchema, perm // predicate is emitted structurally, in the same assembly as every other // clause; policy SQL is never spliced into rendered text afterward, which is // what let a crafted identifier swallow the predicate (#322). - whereParts, whereParams, err := buildWhere(q.Filters, q.TimeRange, bucketSeconds) + whereParts, whereParams, err := buildWhere(q.Filters, q.TimeRange, schema, bucketSeconds) if err != nil { return nil, fmt.Errorf("building WHERE clause: %w", err) } @@ -234,7 +238,7 @@ func validateAndAuthorizeColumns(q *StructuredQuery, colSet map[string]bool, per } // The alias is backtick-quoted by aggregationExpr, so any legal ClickHouse // name is safe against injection. The lone refusal is a '?', which would - // break clickhouse-go's positional value binding. + // shift the positional-to-named parameter rewrite (see NamedParams). if chsql.BindUnsafe(a.Alias) { return fmt.Errorf("unsupported aggregation alias (contains '?'): %s", a.Alias) } @@ -255,7 +259,7 @@ func validateAndAuthorizeColumns(q *StructuredQuery, colSet map[string]bool, per // (ORDER BY an aggregation's AS name). Aliases carry no column policy; // the aggregation that defines them was authorized above. The name is // backtick-quoted when emitted, so any content is safe except a '?' - // (would break value binding, as for aliases). + // (would shift the parameter rewrite, as for aliases). if chsql.BindUnsafe(o.Column) { return fmt.Errorf("unsupported order column (contains '?'): %s", o.Column) } @@ -307,12 +311,12 @@ func resolveProjection(q *StructuredQuery, schema *discovery.TableSchema, perms } } -func buildWhere(filters []Filter, timeRange *TimeRange, bucketSeconds int) ([]string, []any, error) { +func buildWhere(filters []Filter, timeRange *TimeRange, schema *discovery.TableSchema, bucketSeconds int) ([]string, []any, error) { var parts []string var params []any for _, f := range filters { - clause, p, err := filterToSQL(f) + clause, p, err := filterToSQL(f, isDateTimeColumn(schema, f.Column)) if err != nil { return nil, nil, fmt.Errorf("filter on column %q: %w", f.Column, err) } @@ -346,9 +350,12 @@ func buildWhere(filters []Filter, timeRange *TimeRange, bucketSeconds int) ([]st return parts, params, nil } -func filterToSQL(f Filter) (string, []any, error) { +func filterToSQL(f Filter, dateTime bool) (string, []any, error) { col := chsql.QuoteIdent(f.Column) - val := coerceFilterValue(f.Value) + val := f.Value + if dateTime { + val = coerceFilterValue(val) + } switch strings.ToLower(f.Op) { case "eq": return col + " = ?", []any{val}, nil @@ -365,10 +372,18 @@ func filterToSQL(f Filter) (string, []any, error) { case "like": return col + " LIKE ?", []any{val}, nil case "in": + // One placeholder for the whole list, bound as one Array(String) + // parameter rather than a parameter per element, so a long list + // costs one query-string field. if vals, ok := f.Value.([]any); ok && len(vals) > 0 { - placeholders := strings.Repeat("?,", len(vals)) - placeholders = placeholders[:len(placeholders)-1] - return fmt.Sprintf("%s IN (%s)", col, placeholders), vals, nil + if dateTime { + coerced := make([]any, len(vals)) + for i, v := range vals { + coerced[i] = coerceFilterValue(v) + } + vals = coerced + } + return col + " IN ?", []any{vals}, nil } return "", nil, fmt.Errorf("invalid value for 'in' operator") default: @@ -376,11 +391,13 @@ func filterToSQL(f Filter) (string, []any, error) { } } -// coerceFilterValue converts string values that look like RFC3339 timestamps -// to ClickHouse-compatible DateTime strings preserving sub-second precision. -// The clickhouse-go driver's time.Time formatting uses toDateTime() (second -// precision), which loses milliseconds needed for DateTime64 cursor comparisons. -// Returning a formatted string lets ClickHouse parse it with full precision. +// coerceFilterValue rewrites a filter value on a DateTime or DateTime64 column +// that is an RFC3339 timestamp into ClickHouse's own DateTime spelling, in +// UTC, keeping its sub-second digits. ClickHouse reads a String compared +// against such a column with its basic parser, which refuses the RFC3339 +// spelling: measured on 24.8.14.39 and 26.6.3.62, a `T`, a `Z` or an offset +// is TYPE_MISMATCH on DateTime, and on DateTime64 too on 24.8 and inside an +// `in` list on both. The rewritten spelling parses on every one of them. // // A value that isn't a timestamp (a plain string, a number, etc.) is a valid // non-temporal filter value, so the parse "failure" is just the expected @@ -442,8 +459,8 @@ func expandDayWeek(s string) string { // (which the builder surfaces as a 400) rather than returned unchanged: passing // a raw string to ClickHouse surfaces as an opaque DateTime parse error (#285). // -// The output deliberately matches coerceFilterValue's format rather than RFC3339: -// a bare "…T…Z" string is rejected by DateTime64 columns. +// The output deliberately matches coerceFilterValue's format rather than +// RFC3339, which ClickHouse refuses on a DateTime column (see there). func resolveTimeValue(val string, bucketSeconds int) (string, error) { // Try a relative duration first (e.g., "1h", "30m", "7d", "2w"). Go's // time.ParseDuration only understands units up to hours, so day/week @@ -468,6 +485,27 @@ func bucketTime(t time.Time, bucketSeconds int) time.Time { return t.Truncate(d) } +// isDateTimeColumn reports whether the schema types column as DateTime or +// DateTime64, possibly Nullable or LowCardinality. +func isDateTimeColumn(schema *discovery.TableSchema, column string) bool { + c, ok := schema.Lookup(column) + if !ok { + return false + } + t := c.Type + for { + switch { + case strings.HasPrefix(t, "Nullable(") && strings.HasSuffix(t, ")"): + t = t[len("Nullable(") : len(t)-1] + continue + case strings.HasPrefix(t, "LowCardinality(") && strings.HasSuffix(t, ")"): + t = t[len("LowCardinality(") : len(t)-1] + continue + } + return strings.HasPrefix(t, "DateTime") + } +} + func schemaColumnSet(schema *discovery.TableSchema) map[string]bool { m := make(map[string]bool, len(schema.Columns)) for _, c := range schema.Columns { @@ -514,3 +552,154 @@ func isValidAggFn(fn string) bool { } return false } + +// ─── ClickHouse named-parameter binding ───────────────────────────────────── + +// NamedParams rewrites the positional `?` placeholders in the built SQL into +// ClickHouse named parameters and renders each bound value as the text +// ClickHouse will read it back from. It returns the rewritten SQL and the +// values for `param_p0` … `param_pN-1`, positionally. +// +// Every scalar binds as `{pN:String}` and every list as `{pN:Array(String)}`. +// String is not a weaker binding than the column's own type: ClickHouse +// converts the parameter to the column's type for the comparison, so +// `UInt8 = {p:String}` with "256" is false and with "1.5" is a type error, +// matching what the server answers for the same literal. +// +// The rewrite is a left-to-right scan for `?`, which is exact for this SQL and +// only for this SQL: Build never renders a value or a string literal, and +// chsql.BindUnsafe rejects a `?` in any identifier it quotes. +func (r *BuildResult) NamedParams() (string, []string, error) { + if len(r.Params) == 0 { + if strings.Contains(r.SQL, "?") { + return "", nil, fmt.Errorf("query has placeholders but no bound values") + } + return r.SQL, nil, nil + } + + params := make([]string, 0, len(r.Params)) + var b strings.Builder + b.Grow(len(r.SQL) + len(r.Params)*12) + + rest := r.SQL + for i, v := range r.Params { + q := strings.IndexByte(rest, '?') + if q < 0 { + return "", nil, fmt.Errorf("query has %d bound values but only %d placeholders", len(r.Params), i) + } + text, placeholder, err := chParamValue("p"+strconv.Itoa(i), v) + if err != nil { + return "", nil, err + } + b.WriteString(rest[:q]) + b.WriteString(placeholder) + params = append(params, text) + rest = rest[q+1:] + } + if strings.Contains(rest, "?") { + return "", nil, fmt.Errorf("query has more placeholders than the %d bound values", len(r.Params)) + } + b.WriteString(rest) + return b.String(), params, nil +} + +// chParamValue renders one bound value as its ClickHouse query-parameter text +// and the SQL its placeholder becomes, for the parameter called name. +func chParamValue(name string, v any) (text, placeholder string, err error) { + if vals, ok := v.([]any); ok { + lit, err := chArrayLiteral(vals) + if err != nil { + return "", "", err + } + return lit, "{" + name + ":Array(String)}", nil + } + raw, err := chScalarText(v) + if err != nil { + return "", "", err + } + return escapeStringParam(raw), "{" + name + ":String}", nil +} + +// chScalarText is one scalar's value as plain text, before any encoding — +// what the caller means, not what the wire needs. +func chScalarText(v any) (string, error) { + switch val := v.(type) { + case string: + return val, nil + case json.Number: + // The caller's own digits, not a float64 round-trip: 12.50 stays + // "12.50" and an integer past 2^53 keeps every digit. + return val.String(), nil + case bool: + return strconv.FormatBool(val), nil + case float64: + return strconv.FormatFloat(val, 'f', -1, 64), nil + case int: + return strconv.Itoa(val), nil + case int64: + return strconv.FormatInt(val, 10), nil + case uint64: + return strconv.FormatUint(val, 10), nil + case nil: + // `col = NULL` is never true in SQL, so a null used to answer "no + // rows"; an empty String parameter would instead compare against the + // empty string, which is a different question. Refuse it (→ 400) + // rather than answer a question the caller did not ask. + return "", fmt.Errorf("filter value must not be null") + default: + return "", fmt.Errorf("unsupported filter value type %T", v) + } +} + +// chArrayLiteral renders a list as the `['a','b']` text an Array(String) +// query parameter is parsed from. A nested list has no place inside an `in` +// list, and the elements take quoteCHElement's encoding INSTEAD of the +// scalar one, not on top of it. +func chArrayLiteral(vals []any) (string, error) { + var b strings.Builder + b.WriteByte('[') + for i, v := range vals { + if _, isList := v.([]any); isList { + return "", fmt.Errorf("nested list in an 'in' value") + } + text, err := chScalarText(v) + if err != nil { + return "", err + } + if i > 0 { + b.WriteByte(',') + } + b.WriteString(quoteCHElement(text)) + } + b.WriteByte(']') + return b.String(), nil +} + +// quoteCHElement wraps one already-rendered value as a single-quoted element +// of an Array(String) parameter literal. That literal is read as a quoted +// value rather than an escaped field — a raw tab or newline inside the quotes +// round-trips untouched — so only the quote and the backslash need encoding, +// and the scalar encoding must NOT be applied on top of it. +func quoteCHElement(s string) string { + var b strings.Builder + b.Grow(len(s) + 2) + b.WriteByte('\'') + for i := 0; i < len(s); i++ { + if s[i] == '\\' || s[i] == '\'' { + b.WriteByte('\\') + } + b.WriteByte(s[i]) + } + b.WriteByte('\'') + return b.String() +} + +// escapeStringParam encodes one value for a `{p:String}` query parameter, +// which ClickHouse reads with its escaped-text reader: a raw backslash starts +// an escape sequence and a raw tab or newline ends the field. +var escapeStringParam = strings.NewReplacer( + `\`, `\\`, + "\t", `\t`, + "\n", `\n`, + "\r", `\r`, +).Replace diff --git a/internal/query/builder_test.go b/internal/query/builder_test.go index 5c31945d..306df4e7 100644 --- a/internal/query/builder_test.go +++ b/internal/query/builder_test.go @@ -1,6 +1,7 @@ package query import ( + "encoding/json" "fmt" "strings" "testing" @@ -136,8 +137,15 @@ func TestBuild_InFilter(t *testing.T) { } result, err := Build("clicks", sq, testSchema(), nil, 0, DefaultMaxRows) require.NoError(t, err) - assert.Contains(t, result.SQL, "`page` IN (?,?)") - assert.Len(t, result.Params, 2) + // One placeholder for the whole list, bound as one Array(String). + assert.Contains(t, result.SQL, "`page` IN ?") + require.Len(t, result.Params, 1) + assert.Equal(t, []any{"/home", "/about"}, result.Params[0]) + + sql, params, err := result.NamedParams() + require.NoError(t, err) + assert.Contains(t, sql, "`page` IN {p0:Array(String)}") + assert.Equal(t, []string{`['/home','/about']`}, params) } func TestBuild_OrderBy(t *testing.T) { @@ -542,6 +550,58 @@ func TestBuild_FilterWithTimestampValue(t *testing.T) { strVal, isString := result.Params[0].(string) assert.True(t, isString, "timestamp filter value should be coerced to formatted string, got %T", result.Params[0]) assert.Equal(t, "2026-04-02 16:02:07.666", strVal) + + sql, params, err := result.NamedParams() + require.NoError(t, err) + assert.Contains(t, sql, "`received_timestamp` < {p0:String}") + assert.Equal(t, []string{"2026-04-02 16:02:07.666"}, params) +} + +// TestBuild_TimestampRewriteIsForDateTimeColumnsOnly: an RFC3339 value is +// rewritten only where the schema types the column DateTime or DateTime64 — +// wrapped or not, scalar or inside an `in` list. The same text on any other +// column is the caller's own value and reaches ClickHouse as written. +func TestBuild_TimestampRewriteIsForDateTimeColumnsOnly(t *testing.T) { + t.Parallel() + schema := &discovery.TableSchema{Name: "events", Columns: []discovery.Column{ + {Name: "dt", Type: "DateTime"}, + {Name: "dtz", Type: "DateTime('Europe/Berlin')"}, + {Name: "dt64", Type: "Nullable(DateTime64(3, 'UTC'))"}, + {Name: "lc", Type: "LowCardinality(Nullable(DateTime))"}, + {Name: "label", Type: "String"}, + {Name: "day", Type: "Date"}, + }} + const rfc = "2026-04-02T16:02:07.666Z" + const ch = "2026-04-02 16:02:07.666" + tests := []struct { + column string + value any + want any + }{ + {"dt", rfc, ch}, + {"dtz", rfc, ch}, + {"dt64", rfc, ch}, + {"lc", rfc, ch}, + {"dt", "2026-04-02 16:02:07", "2026-04-02 16:02:07"}, + {"label", rfc, rfc}, + {"day", rfc, rfc}, + {"dt64", []any{rfc, "2026-04-02 16:02:07"}, []any{ch, "2026-04-02 16:02:07"}}, + {"label", []any{rfc}, []any{rfc}}, + } + for _, tt := range tests { + t.Run(fmt.Sprintf("%s %v", tt.column, tt.value), func(t *testing.T) { + t.Parallel() + op := "eq" + if _, ok := tt.value.([]any); ok { + op = "in" + } + sq := &StructuredQuery{SelectAll: true, Filters: []Filter{{Column: tt.column, Op: op, Value: tt.value}}} + result, err := Build("events", sq, schema, nil, 0, DefaultMaxRows) + require.NoError(t, err) + require.Len(t, result.Params, 1) + assert.Equal(t, tt.want, result.Params[0]) + }) + } } func TestBuild_TableNameWithBacktick(t *testing.T) { @@ -1002,3 +1062,128 @@ func TestBuild_InsertResolvedGrantIsRejected(t *testing.T) { require.NoError(t, err) assert.NotNil(t, res) } + +// TestNamedParams pins the ClickHouse binding rule: every positional `?` +// becomes a named parameter, every scalar binds as String and every list as +// Array(String), in the order the WHERE assembly emitted them. +func TestNamedParams(t *testing.T) { + t.Parallel() + tests := []struct { + name string + sql string + params []any + wantSQL string + wantParams []string + }{ + { + name: "no parameters", + sql: "SELECT `page` FROM `clicks` LIMIT 10", + wantSQL: "SELECT `page` FROM `clicks` LIMIT 10", + wantParams: nil, + }, + { + name: "policy predicate keeps its leading position", + sql: "SELECT `page` FROM `clicks` WHERE (`org_id` = ?) AND `page` = ? LIMIT 100", + params: []any{"org-1", "/home"}, + wantSQL: "SELECT `page` FROM `clicks` WHERE (`org_id` = {p0:String}) AND `page` = {p1:String} LIMIT 100", + wantParams: []string{"org-1", "/home"}, + }, + { + name: "list binds as one Array(String)", + sql: "SELECT * FROM `t` WHERE `page` IN ? LIMIT 10", + params: []any{[]any{"/a", "/b"}}, + wantSQL: "SELECT * FROM `t` WHERE `page` IN {p0:Array(String)} LIMIT 10", + wantParams: []string{`['/a','/b']`}, + }, + { + name: "numbers keep the caller's own digits", + sql: "SELECT * FROM `t` WHERE `a` = ? AND `b` = ? AND `c` = ? LIMIT 10", + params: []any{json.Number("12.50"), json.Number("9007199254740993"), true}, + wantSQL: "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` = {p1:String} " + + "AND `c` = {p2:String} LIMIT 10", + wantParams: []string{"12.50", "9007199254740993", "true"}, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + sql, params, err := (&BuildResult{SQL: tt.sql, Params: tt.params}).NamedParams() + require.NoError(t, err) + assert.Equal(t, tt.wantSQL, sql) + assert.Equal(t, tt.wantParams, params) + }) + } +} + +// TestNamedParams_Encoding pins the two escapings a ClickHouse query parameter +// needs, which are NOT the same and must not be applied to each other's +// values. Measured on 26.6.3.62 (the version the integration suite pins): +// +// - a scalar `{p:String}` is read by the escaped-text reader, so a raw +// backslash is taken as the start of an escape sequence ("a\b" came back +// holding a backspace) and a raw tab or newline ends the field outright +// (code 457, a 500 for the caller); +// - an `Array(String)` value is an array literal whose elements are quoted, +// so a raw tab or newline rides through untouched and only the quote and +// the backslash need encoding. +// +// Both encodings round-trip every case below byte for byte against a live +// server; getting either wrong is silent data loss, not an error. +func TestNamedParams_Encoding(t *testing.T) { + t.Parallel() + tests := []struct { + name string + value any + wantParam string + }{ + {"plain", "hello", "hello"}, + {"single quote needs nothing", "it's", "it's"}, + {"backslash", `a\b`, `a\\b`}, + {"windows path", `C:\Users\x`, `C:\\Users\\x`}, + {"tab", "a\tb", `a\tb`}, + {"newline", "a\nb", `a\nb`}, + {"carriage return", "a\rb", `a\rb`}, + {"a literal backslash-n", `a\nb`, `a\\nb`}, + {"like pattern", "%foo%", "%foo%"}, + {"list quotes and backslashes", []any{`it's`, `a\b`, "a\tb"}, "['it\\'s','a\\\\b','a\tb']"}, + {"list containment attempt", []any{`']) OR 1=1 --`}, `['\']) OR 1=1 --']`}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + _, params, err := (&BuildResult{SQL: "SELECT ?", Params: []any{tt.value}}).NamedParams() + require.NoError(t, err) + require.Len(t, params, 1) + assert.Equal(t, tt.wantParam, params[0]) + }) + } +} + +// TestNamedParams_Rejects covers the values and shapes that have no honest +// binding. A JSON null is the notable one: the driver quietly turned it into +// `col = NULL` (never true), where an empty String parameter would compare +// against the empty string — a different question, so it is refused (→ 400). +func TestNamedParams_Rejects(t *testing.T) { + t.Parallel() + tests := []struct { + name string + sql string + params []any + wantErr string + }{ + {"null value", "SELECT ?", []any{nil}, "must not be null"}, + {"object value", "SELECT ?", []any{map[string]any{"k": "v"}}, "unsupported filter value type"}, + {"nested list", "SELECT ?", []any{[]any{[]any{"a"}}}, "nested list"}, + {"more values than placeholders", "SELECT 1", []any{"a"}, "only 0 placeholders"}, + {"more placeholders than values", "SELECT ?, ?", []any{"a"}, "more placeholders"}, + {"placeholder with no values", "SELECT ?", nil, "no bound values"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + _, _, err := (&BuildResult{SQL: tt.sql, Params: tt.params}).NamedParams() + require.Error(t, err) + assert.Contains(t, err.Error(), tt.wantErr) + }) + } +} From 6dd65fb838d6afcba4dce29fc0edb855376e2157 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:53:05 -0400 Subject: [PATCH 07/70] feat(typelayer): parse a row under the INSERT column list it was published with ParseRow accepted only the generation's full wire column list, so a row published by a role that may write some columns could never be read. It now accepts any duplicate-free subset of the wire columns, in any order, and names that list on the parse (chtypes.WithColumns), so every unlisted column holds the DEFAULT the server would store for it, computed with the listed values in scope. A list naming an unknown or repeated column is still ErrColumnsDrift. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/typelayer/errors.go | 7 +-- internal/typelayer/filter.go | 52 +++++++++++++++--- internal/typelayer/filter_test.go | 87 ++++++++++++++++++++++++++++--- 3 files changed, 129 insertions(+), 17 deletions(-) diff --git a/internal/typelayer/errors.go b/internal/typelayer/errors.go index 14867bc9..d8292f39 100644 --- a/internal/typelayer/errors.go +++ b/internal/typelayer/errors.go @@ -39,7 +39,8 @@ func IsUnavailable(err error) bool { } // ErrColumnsDrift is returned by ParseRow when the envelope's column list is -// not the one the current compiled handle exports. A positional row is only -// interpretable against the generation that produced it, so a mismatch means -// the event predates a schema change and must be withheld rather than guessed. +// not an INSERT column list the current compiled handle accepts: it names a +// column the handle does not export, or names one twice. The event predates a +// schema change (or was not written by this gateway) and must be withheld +// rather than read under guessed positions. var ErrColumnsDrift = errors.New("row columns do not match the table's wire columns") diff --git a/internal/typelayer/filter.go b/internal/typelayer/filter.go index 14cce201..e57e8ae2 100644 --- a/internal/typelayer/filter.go +++ b/internal/typelayer/filter.go @@ -48,16 +48,27 @@ type Row struct { } // ParseRow parses one JSONCompactEachRow line, with or without its trailing -// newline. columns must be the generation's wire columns exactly: a positional -// row is uninterpretable against any other order, so a mismatch is -// ErrColumnsDrift rather than a guess. +// newline, whose k-th field belongs to the k-th name in columns: the column +// list the ingest worker's INSERT names for the same line. +// +// columns is the generation's wire columns, or any duplicate-free subset of +// them in any order, because a role that may write only some columns publishes +// only those. The parse names that list (chtypes.WithColumns, the INSERT +// column list), so an unlisted column holds what the server stores for it: its +// DEFAULT, computed with the listed values in scope. A name the generation +// does not export (a column dropped since, a computed column, a typo), a +// repeated name or an empty list is ErrColumnsDrift: no INSERT could store +// that row in this table. func (t *Table) ParseRow(columns []string, row []byte) (*Row, error) { if t.pool == nil { return nil, &Unavailable{Tenant: t.tenant, Table: t.Name, Cause: t.cause} } + var opts []chtypes.RowOption if !slices.Equal(columns, t.WireColumns) { - return nil, fmt.Errorf("%w: event carries %v, generation %d exports %v", - ErrColumnsDrift, columns, t.Generation, t.WireColumns) + if err := t.insertableList(columns); err != nil { + return nil, err + } + opts = []chtypes.RowOption{chtypes.WithColumns(columns)} } body := row if n := len(body); n == 0 || body[n-1] != '\n' { @@ -68,7 +79,7 @@ func (t *Table) ParseRow(columns []string, row []byte) (*Row, error) { // serialization the pool exists to avoid. p := t.pool s := p.acquire() - block, err := s.schema.ParseBlock(chtypes.JSONCompactEachRow, body, InsertSettings()) + block, err := s.schema.ParseBlock(chtypes.JSONCompactEachRow, body, InsertSettings(), opts...) if err != nil { p.release(s) return nil, err @@ -76,6 +87,35 @@ func (t *Table) ParseRow(columns []string, row []byte) (*Row, error) { return &Row{table: t, pool: p, slot: s, block: block}, nil } +// insertableList reports why columns cannot be an INSERT column list over +// this generation's wire columns, nil when it can. Checked here rather than +// left to the server's own refusal (codes 16 and 15) so a caller can tell +// drift from a row that does not parse. +func (t *Table) insertableList(columns []string) error { + drift := func(why string) error { + return fmt.Errorf("%w: event carries %v (%s), generation %d exports %v", + ErrColumnsDrift, columns, why, t.Generation, t.WireColumns) + } + if len(columns) == 0 { + return drift("no columns") + } + listed := make(map[string]bool, len(t.WireColumns)) + for _, c := range t.WireColumns { + listed[c] = false + } + for _, c := range columns { + seen, known := listed[c] + switch { + case !known: + return drift(fmt.Sprintf("%q is not one of them", c)) + case seen: + return drift(fmt.Sprintf("%q is repeated", c)) + } + listed[c] = true + } + return nil +} + // Close frees the parsed block and gives its handle back. Required: the C // layer does not refcount. A second Close is a no-op. func (r *Row) Close() { diff --git a/internal/typelayer/filter_test.go b/internal/typelayer/filter_test.go index fe1277ec..8383e5ab 100644 --- a/internal/typelayer/filter_test.go +++ b/internal/typelayer/filter_test.go @@ -424,21 +424,92 @@ func TestVisibleWithReason_LabelsTheCause(t *testing.T) { assert.Nil(t, tbl.filterOn(row.slot, "notAFunction(`tenant`) = {p0:String}", map[string]string{"p0": "x"})) } -// TestParseRow_ColumnsDriftIsAnError: a positional row is only interpretable -// against the generation that produced it. +// TestParseRow_ColumnsDriftIsAnError: a column list no INSERT into this +// generation could name is drift, not a parse attempt. A row whose list is +// fine but whose values do not fit it is not drift: the block holds the row's +// refusal and every predicate over it withholds (measured on the 26.6 +// artifact: ParseBlock reports no call-level error for a malformed row). func TestParseRow_ColumnsDriftIsAnError(t *testing.T) { eng := TestEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) defer tbl.Release() - _, err = tbl.ParseRow([]string{"id", "tenant"}, []byte(sampleRow)) - require.ErrorIs(t, err, ErrColumnsDrift) + for name, cols := range map[string][]string{ + "unknown column": {"id", "dropped"}, + "case differs": {"id", "Tenant"}, + "repeated column": {"id", "id"}, + "empty list": {}, + "nil list": nil, + "full list, one extra": append(append([]string(nil), tbl.WireColumns...), "extra"), + } { + _, err = tbl.ParseRow(cols, []byte(`[7, "x"]`)) + require.ErrorIs(t, err, ErrColumnsDrift, name) + } + + row, err := tbl.ParseRow([]string{"id", "tenant"}, []byte(sampleRow)) + require.NoError(t, err, "a valid list with a bad row is not drift") + defer row.Close() + ok, reason := row.VisibleWithReason([]Predicate{{Column: "id", Op: "=", Values: []string{"7"}}}) + assert.False(t, ok, "seven fields under a two-column list do not parse, so nothing matches") + assert.Equal(t, ReasonDecline, reason) +} + +// defaultsTable has the column kinds an INSERT column list interacts with: a +// literal DEFAULT, a Nullable column with a DEFAULT, and a DEFAULT computed +// from another column. +func defaultsTable() *discovery.TableSchema { + return &discovery.TableSchema{ + Name: "events", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt32", Position: 1}, + {Name: "tenant", Type: "String", Position: 2}, + {Name: "region", Type: "String", Position: 3, DefaultKind: "DEFAULT", DefaultExpression: "'eu'"}, + {Name: "note", Type: "Nullable(String)", Position: 4, DefaultKind: "DEFAULT", DefaultExpression: "'n/a'"}, + {Name: "next", Type: "UInt32", Position: 5, DefaultKind: "DEFAULT", DefaultExpression: "id + 1"}, + }, + } +} - reordered := append([]string(nil), tbl.WireColumns...) - reordered[0], reordered[1] = reordered[1], reordered[0] - _, err = tbl.ParseRow(reordered, []byte(sampleRow)) - require.ErrorIs(t, err, ErrColumnsDrift, "same names in a different order is still drift") +// TestParseRow_ColumnSubsetTakesTheServersDefaults: a narrower list, in any +// order, parses as the INSERT naming those columns would store the row. Every +// unlisted column holds its DEFAULT — a Nullable one too, which a null-padded +// full-width row would have stored as NULL — and a DEFAULT over a listed +// column is computed from the listed value. +func TestParseRow_ColumnSubsetTakesTheServersDefaults(t *testing.T) { + eng := TestEngine(t, defaultsTable()) + tbl, err := eng.Table(tenant.Default, "events") + require.NoError(t, err) + defer tbl.Release() + + for name, tc := range map[string]struct { + cols []string + line string + }{ + "declaration order": {[]string{"id", "tenant"}, `[5, "acme"]`}, + "permuted": {[]string{"tenant", "id"}, `["acme", 5]`}, + } { + row, err := tbl.ParseRow(tc.cols, []byte(tc.line)) + require.NoError(t, err, name) + for _, p := range []Predicate{ + {Column: "id", Op: "=", Values: []string{"5"}}, + {Column: "tenant", Op: "=", Values: []string{"acme"}}, + {Column: "region", Op: "=", Values: []string{"eu"}}, + {Column: "note", Op: "=", Values: []string{"n/a"}}, + {Column: "next", Op: "=", Values: []string{"6"}}, + } { + ok, reason := row.VisibleWithReason([]Predicate{p}) + assert.True(t, ok, "%s: %s %s %v: %s", name, p.Column, p.Op, p.Values, reason) + } + row.Close() + } + + // The full list still takes the no-list path and reads every field. + row, err := tbl.ParseRow(tbl.WireColumns, []byte(`[5, "acme", "us", null, 9]`)) + require.NoError(t, err) + defer row.Close() + assert.True(t, row.Visible([]Predicate{{Column: "region", Op: "=", Values: []string{"us"}}})) + assert.True(t, row.Visible([]Predicate{{Column: "next", Op: "=", Values: []string{"9"}}})) } func TestParseRow_AcceptsALineWithOrWithoutNewline(t *testing.T) { From 35a008ed8f50021f602cd96e3c179048e5e34561 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:53:57 -0400 Subject: [PATCH 08/70] feat(query): bind integer policy claims through the strict cast The structured query's row filter now comes from WhereSQL with the discovered column types: a policy claim on an integer column (any width, Nullable or LowCardinality) binds as one {pN:String} parameter compared through the strict round-trip cast, so a value that is not the canonical spelling of an in-range integer matches nothing instead of wrapping. Every other value keeps the plain {pN:String} binding, encoded by chsql.EscapeStringParam, the same encoding the type layer's row filters use. The e2e suite pins the rendering of a Decimal and a DateTime64, RFC 3339 filter values on DateTime and DateTime64 columns, a timestamp read back from /v1/query filtering as it is, and the 400 for a null filter value. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/api/cache_tenant_test.go | 2 +- internal/api/pipes_test.go | 6 +- internal/query/builder.go | 58 ++++++----- internal/query/builder_test.go | 165 ++++++++++++++++++++++++++++-- tests/e2e/sdk/query.test.ts | 104 +++++++++++++++++++ 5 files changed, 298 insertions(+), 37 deletions(-) diff --git a/internal/api/cache_tenant_test.go b/internal/api/cache_tenant_test.go index 16ca37f7..59b3175e 100644 --- a/internal/api/cache_tenant_test.go +++ b/internal/api/cache_tenant_test.go @@ -165,7 +165,7 @@ func TestNewRouter_FlatDirectoryCacheStillHits(t *testing.T) { // one tenant. The singleflight key is the tenant-led cache key, so two // tenants' identical requests in flight together are two queries: neither // waits on, or receives, the other's result. Under synctest the count is -// exact: Wait returns once every request is either inside Query or parked on +// exact: Wait returns once every request is either inside ClickHouse or parked on // another's flight, with no sleep to race. func TestCachedRoutes_SingleflightIsPerTenant(t *testing.T) { tenants := nestedTenants(t, map[string]string{"acme": fullConfig(100), "globex": fullConfig(200)}) diff --git a/internal/api/pipes_test.go b/internal/api/pipes_test.go index 036f0cc9..acc54410 100644 --- a/internal/api/pipes_test.go +++ b/internal/api/pipes_test.go @@ -366,7 +366,7 @@ func TestPipesHandler_Execute_ParamsFromQuery(t *testing.T) { safeHandle(h.Execute, w, withTenant(r)) - // Should pass param binding — will fail later at executeQuery (nil conn). + // Should pass param binding — failing later at the unwired target. assert.NotEqual(t, http.StatusBadRequest, w.Code) assert.NotEqual(t, http.StatusNotFound, w.Code) } @@ -420,7 +420,7 @@ func TestPipesHandler_Execute_PostBodyParams(t *testing.T) { safeHandle(h.Execute, w, withTenant(r)) - // Should pass param binding — will fail at executeQuery (nil conn). + // Should pass param binding — failing later at the unwired target. assert.NotEqual(t, http.StatusBadRequest, w.Code) assert.NotEqual(t, http.StatusNotFound, w.Code) } @@ -572,7 +572,7 @@ func TestPipesHandler_Execute_MutationRunsEveryCall(t *testing.T) { // Identical mutation calls in flight together are each executed: coalescing // them would run one write for all of them. Under synctest, Wait returns once -// every request is inside Exec or parked on another's flight. +// every request is inside ClickHouse or parked on another's flight. func TestPipesHandler_Execute_ConcurrentMutationsNotCoalesced(t *testing.T) { synctest.Test(t, func(t *testing.T) { const calls = 3 diff --git a/internal/query/builder.go b/internal/query/builder.go index 49aa6e85..e1aecab8 100644 --- a/internal/query/builder.go +++ b/internal/query/builder.go @@ -119,13 +119,15 @@ func Build(table string, q *StructuredQuery, schema *discovery.TableSchema, perm // - aggregations only → validateAndAuthorizeColumns → IsAggregationAllowed // // All three fail closed on a nil Select; all three are pinned by - // TestBuild_InsertResolvedGrantIsRejected. The bare read is the backstop if + // TestBuild_InsertResolvedGrantIsRejected. The bare call is the backstop if // that ordering changes — it would panic rather than skip the filter. Do NOT // add a `perms.Select != nil` guard: it would emit an unfiltered query, the // silent widening the pointer shape exists to prevent. - if perms != nil && perms.Select.WhereClause != "" { - whereParts = append([]string{"(" + perms.Select.WhereClause + ")"}, whereParts...) - params = append(params, perms.Select.WhereParams...) + if perms != nil { + if clause, rowParams := perms.Select.WhereSQL(columnType(schema)); clause != "" { + whereParts = append([]string{"(" + clause + ")"}, whereParts...) + params = append(params, rowParams...) + } } params = append(params, whereParams...) if len(whereParts) > 0 { @@ -485,6 +487,15 @@ func bucketTime(t time.Time, bucketSeconds int) time.Time { return t.Truncate(d) } +// columnType looks a column's ClickHouse type up in the discovered schema, so +// the row filter can bind an integer column's claims through the strict cast. +func columnType(schema *discovery.TableSchema) func(string) string { + return func(col string) string { + c, _ := schema.Lookup(col) + return c.Type + } +} + // isDateTimeColumn reports whether the schema types column as DateTime or // DateTime64, possibly Nullable or LowCardinality. func isDateTimeColumn(schema *discovery.TableSchema, column string) bool { @@ -560,11 +571,15 @@ func isValidAggFn(fn string) bool { // ClickHouse will read it back from. It returns the rewritten SQL and the // values for `param_p0` … `param_pN-1`, positionally. // -// Every scalar binds as `{pN:String}` and every list as `{pN:Array(String)}`. -// String is not a weaker binding than the column's own type: ClickHouse -// converts the parameter to the column's type for the comparison, so -// `UInt8 = {p:String}` with "256" is false and with "1.5" is a type error, -// matching what the server answers for the same literal. +// Every scalar binds as `{pN:String}` and every list as `{pN:Array(String)}` +// — one rule, shared with the row filters the type layer compiles. String is +// not a weaker binding than the column's own type: ClickHouse converts the +// parameter to the column's type for the comparison, so `UInt8 = {p:String}` +// with "256" is false and with "1.5" is a type error, matching what the +// server answers for the same literal. The exception is a policy claim on an +// integer column (a chsql.IntParam): its placeholder expands to +// chsql.StrictInt over the one `{pN:String}` parameter, because the plain +// form wraps a value at or past 2^64. // // The rewrite is a left-to-right scan for `?`, which is exact for this SQL and // only for this SQL: Build never renders a value or a string literal, and @@ -606,18 +621,21 @@ func (r *BuildResult) NamedParams() (string, []string, error) { // chParamValue renders one bound value as its ClickHouse query-parameter text // and the SQL its placeholder becomes, for the parameter called name. func chParamValue(name string, v any) (text, placeholder string, err error) { - if vals, ok := v.([]any); ok { - lit, err := chArrayLiteral(vals) + switch val := v.(type) { + case []any: + lit, err := chArrayLiteral(val) if err != nil { return "", "", err } return lit, "{" + name + ":Array(String)}", nil + case chsql.IntParam: + return chsql.EscapeStringParam(val.Value), chsql.StrictInt(name, val.Type), nil } raw, err := chScalarText(v) if err != nil { return "", "", err } - return escapeStringParam(raw), "{" + name + ":String}", nil + return chsql.EscapeStringParam(raw), "{" + name + ":String}", nil } // chScalarText is one scalar's value as plain text, before any encoding — @@ -653,8 +671,8 @@ func chScalarText(v any) (string, error) { // chArrayLiteral renders a list as the `['a','b']` text an Array(String) // query parameter is parsed from. A nested list has no place inside an `in` -// list, and the elements take quoteCHElement's encoding INSTEAD of the -// scalar one, not on top of it. +// list, and the elements take quoteCHElement's encoding INSTEAD of +// chsql.EscapeStringParam's, not on top of it. func chArrayLiteral(vals []any) (string, error) { var b strings.Builder b.WriteByte('[') @@ -679,7 +697,7 @@ func chArrayLiteral(vals []any) (string, error) { // of an Array(String) parameter literal. That literal is read as a quoted // value rather than an escaped field — a raw tab or newline inside the quotes // round-trips untouched — so only the quote and the backslash need encoding, -// and the scalar encoding must NOT be applied on top of it. +// and chsql.EscapeStringParam's encoding must NOT be applied on top of it. func quoteCHElement(s string) string { var b strings.Builder b.Grow(len(s) + 2) @@ -693,13 +711,3 @@ func quoteCHElement(s string) string { b.WriteByte('\'') return b.String() } - -// escapeStringParam encodes one value for a `{p:String}` query parameter, -// which ClickHouse reads with its escaped-text reader: a raw backslash starts -// an escape sequence and a raw tab or newline ends the field. -var escapeStringParam = strings.NewReplacer( - `\`, `\\`, - "\t", `\t`, - "\n", `\n`, - "\r", `\r`, -).Replace diff --git a/internal/query/builder_test.go b/internal/query/builder_test.go index 306df4e7..0978c8da 100644 --- a/internal/query/builder_test.go +++ b/internal/query/builder_test.go @@ -186,10 +186,18 @@ func TestBuild_TimeRange(t *testing.T) { assert.Len(t, result.Params, 1) } -// permsWithFilter returns resolved permissions carrying a row-filter predicate, -// shaped exactly as policy.Evaluate emits one (quoted column, positional '?'). +// permsWithFilter returns resolved permissions carrying a row-filter predicate +// on org_id (a String column), resolved by policy.Evaluate itself. func permsWithFilter() *policy.ResolvedPermissions { - return &policy.ResolvedPermissions{Allowed: true, Select: &policy.ResolvedSelect{WhereClause: "`org_id` = ?", WhereParams: []any{"org-1"}}} + return permsFiltering(map[string]policy.Filter{"org_id": {Eq: new("org-1")}}, nil) +} + +// permsFiltering resolves a read grant carrying filter for role "r" on clicks. +func permsFiltering(filter map[string]policy.Filter, claims map[string]any) *policy.ResolvedPermissions { + p := &policy.Policy{Tables: map[string]policy.TablePolicy{ + "clicks": {"r": {Select: &policy.SelectPermissions{Filter: filter}}}, + }} + return policy.Evaluate(p, "r", "clicks", "select", claims) } // TestBuild_PolicyPredicate pins the structural emission of the row-level- @@ -267,6 +275,147 @@ func TestBuild_PolicyPredicate_SurvivesCraftedIdentifiers(t *testing.T) { } } +// TestBuild_PolicyPredicate_IntegerColumnsBindThroughTheStrictCast pins the +// query path's half of the integer-claim rule: a policy claim compared against +// an integer column (any width, Nullable or LowCardinality) renders as +// chsql.StrictInt over ONE {pN:String} parameter, on every operator and on +// each element of an _in list, while every other column keeps the plain +// {pN:String} form. The typelayer renders the same expression for the stream +// and the insert check. +func TestBuild_PolicyPredicate_IntegerColumnsBindThroughTheStrictCast(t *testing.T) { + t.Parallel() + schema := &discovery.TableSchema{Name: "clicks", Columns: []discovery.Column{ + {Name: "page", Type: "String"}, + {Name: "u64", Type: "UInt64"}, + {Name: "i8", Type: "Int8"}, + {Name: "nu256", Type: "Nullable(UInt256)"}, + {Name: "lci32", Type: "LowCardinality(Nullable(Int32))"}, + {Name: "org_id", Type: "String"}, + {Name: "amount", Type: "Decimal(18, 4)"}, + {Name: "flag", Type: "Bool"}, + }} + e := func(p, typ string) string { + c := "accurateCastOrNull({" + p + ":String}, '" + typ + "')" + return "if(toString(" + c + ") = {" + p + ":String}, " + c + ", NULL)" + } + tests := []struct { + name string + column string + filter policy.Filter + claims map[string]any + wantWhere string + wantParams []string + }{ + { + "eq on UInt64", "u64", + policy.Filter{Eq: new("{{ jwt.t }}")}, + map[string]any{"t": "5"}, + "`u64` = " + e("p0", "UInt64"), + []string{"5"}, + }, + { + "neq on Int8", "i8", + policy.Filter{Neq: new("-3")}, + nil, + "`i8` != " + e("p0", "Int8"), + []string{"-3"}, + }, + { + "gt on Nullable(UInt256) casts to the bare type", "nu256", + policy.Filter{Gt: new("7")}, + nil, + "`nu256` > " + e("p0", "UInt256"), + []string{"7"}, + }, + { + "lt on LowCardinality(Nullable(Int32)) casts to the bare type", "lci32", + policy.Filter{Lt: new("9")}, + nil, + "`lci32` < " + e("p0", "Int32"), + []string{"9"}, + }, + { + "in on UInt64 casts each element", "u64", + policy.Filter{In: new("{{ jwt.ts }}")}, + map[string]any{"ts": []any{"5", "18446744073709551621", "007"}}, + "`u64` IN (" + e("p0", "UInt64") + "," + e("p1", "UInt64") + "," + e("p2", "UInt64") + ")", + []string{"5", "18446744073709551621", "007"}, + }, + { + "a claim needing escape is encoded once", "u64", + policy.Filter{Eq: new("{{ jwt.t }}")}, + map[string]any{"t": `a\b`}, + "`u64` = " + e("p0", "UInt64"), + []string{`a\\b`}, + }, + { + "eq on String keeps the plain form", "org_id", + policy.Filter{Eq: new("acme")}, + nil, + "`org_id` = {p0:String}", + []string{"acme"}, + }, + { + "in on String keeps the plain form", "org_id", + policy.Filter{In: new("{{ jwt.ts }}")}, + map[string]any{"ts": []any{"a", "b"}}, + "`org_id` IN ({p0:String},{p1:String})", + []string{"a", "b"}, + }, + { + "Decimal keeps the plain form", "amount", + policy.Filter{Gt: new("1.50")}, + nil, + "`amount` > {p0:String}", + []string{"1.50"}, + }, + { + "Bool keeps the plain form", "flag", + policy.Filter{Eq: new("true")}, + nil, + "`flag` = {p0:String}", + []string{"true"}, + }, + { + "an unresolvable claim still fails closed", "u64", + policy.Filter{Eq: new("{{ jwt.absent }}")}, + nil, + "1 = 0", nil, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + perms := permsFiltering(map[string]policy.Filter{tt.column: tt.filter}, tt.claims) + res, err := Build("clicks", &StructuredQuery{Columns: []string{"page"}}, schema, perms, 0, DefaultMaxRows) + require.NoError(t, err) + sql, params, err := res.NamedParams() + require.NoError(t, err) + assert.Equal(t, "SELECT `page` FROM `clicks` WHERE ("+tt.wantWhere+") LIMIT 10000", sql) + assert.Equal(t, tt.wantParams, params) + }) + } +} + +// TestBuild_PolicyPredicate_CallerFiltersKeepThePlainForm: the strict cast is +// for policy claims only. A caller's own filter on an integer column binds as +// before — it can only narrow what the policy already admits. +func TestBuild_PolicyPredicate_CallerFiltersKeepThePlainForm(t *testing.T) { + t.Parallel() + perms := permsFiltering(map[string]policy.Filter{"count": {Eq: new("5")}}, nil) + sq := &StructuredQuery{Columns: []string{"page"}, Filters: []Filter{ + {Column: "count", Op: "gt", Value: json.Number("1")}, + {Column: "count", Op: "in", Value: []any{json.Number("1"), json.Number("2")}}, + }} + res, err := Build("clicks", sq, testSchema(), perms, 0, DefaultMaxRows) + require.NoError(t, err) + sql, params, err := res.NamedParams() + require.NoError(t, err) + assert.Equal(t, "SELECT `page` FROM `clicks` WHERE (`count` = "+chsql.StrictInt("p0", "UInt64")+ + ") AND `count` > {p1:String} AND `count` IN {p2:Array(String)} LIMIT 10000", sql) + assert.Equal(t, []string{"5", "1", "['1','2']"}, params) +} + // TestBuild_PolicyMaxRows pins the role's max_rows cap folded into Build's LIMIT // computation (#322): the emitted LIMIT is min(caller limit, default cap, policy // cap), with non-positive values meaning "no cap from that source". Includes a @@ -633,8 +782,8 @@ func TestBuild_InvalidColumns(t *testing.T) { sq: &StructuredQuery{ Columns: []string{"page"}, // A non-schema order column is allowed as an alias reference and - // backtick-quoted; only a '?' (which clickhouse-go's binder would - // miscount) is rejected. + // backtick-quoted; only a '?' (which the positional-to-named rewrite + // would miscount) is rejected. OrderBy: []OrderClause{{Column: "we?ird", Dir: "asc"}}, }, wantErr: "unsupported order column", @@ -915,7 +1064,7 @@ func TestBuild_AggregationAliasQuotedAndContained(t *testing.T) { } // TestBuild_RejectsBindUnsafeAlias keeps the one alias rejection that remains: a -// '?' would be miscounted by clickhouse-go's positional value binder. +// '?' would be miscounted by the positional-to-named parameter rewrite. func TestBuild_RejectsBindUnsafeAlias(t *testing.T) { t.Parallel() sq := &StructuredQuery{Aggregations: []Aggregation{{Fn: "count", Column: "*", Alias: "we?ird"}}} @@ -1160,8 +1309,8 @@ func TestNamedParams_Encoding(t *testing.T) { } // TestNamedParams_Rejects covers the values and shapes that have no honest -// binding. A JSON null is the notable one: the driver quietly turned it into -// `col = NULL` (never true), where an empty String parameter would compare +// binding. A JSON null is the notable one: it used to bind as `col = NULL` +// (never true), where an empty String parameter would compare // against the empty string — a different question, so it is refused (→ 400). func TestNamedParams_Rejects(t *testing.T) { t.Parallel() diff --git a/tests/e2e/sdk/query.test.ts b/tests/e2e/sdk/query.test.ts index f8ad9fec..764871c4 100644 --- a/tests/e2e/sdk/query.test.ts +++ b/tests/e2e/sdk/query.test.ts @@ -1,3 +1,4 @@ +import type { FilterOp } from "@wavehouse/sdk"; import { beforeAll, describe, expect, it } from "vitest"; import { adminClient, @@ -236,6 +237,109 @@ describe("Query", () => { } }); + // The value SPELLING is part of the SDK contract, and none of the suite + // tables can show it: they are all String/UInt32/DateTime64. A Decimal + // arrives as a JSON number (`wavehouse codegen` types it as one), and a + // DateTime64 keeps ClickHouse's own `YYYY-MM-DD HH:MM:SS.fff` spelling + // rather than ISO-8601 — the same bytes the stream carries (#372). + it("renders a Decimal as a number and a DateTime64 in ClickHouse's spelling", async () => { + const admin = adminClient(); + const t = `types_${testId().replace(/-/g, "_")}`; + + await chQuery( + `CREATE TABLE IF NOT EXISTS default.\`${t}\` (id String, amount Decimal(10, 2), at DateTime64(3)) ENGINE = Memory`, + ); + await chQuery(`INSERT INTO default.\`${t}\` VALUES ('r1', 12.50, '2026-01-15 10:30:00.123')`); + await admin.schema.refresh(); + + const currentPolicy = readPolicyFile(); + await setPolicy({ + tables: { + ...currentPolicy.tables, + [t]: { viewer: { select: { allow_columns: ["*"] } } }, + }, + }); + + try { + const result = await wh.from(t).selectAll().fetch(); + expect(result.error).toBeNull(); + expect(result.data).toHaveLength(1); + const row = result.data![0] as Record; + expect(typeof row.amount).toBe("number"); + expect(row.amount).toBe(12.5); + expect(row.at).toBe("2026-01-15 10:30:00.123"); + } finally { + await setPolicy(currentPolicy); + await chQuery(`DROP TABLE IF EXISTS default.\`${t}\``); + } + }); + + // ClickHouse refuses an RFC 3339 String against a DateTime column, so the + // builder rewrites one on DateTime and DateTime64 columns; a timestamp read + // back from /v1/query, in ClickHouse's own spelling, filters as it is. + it("filters DateTime and DateTime64 columns by RFC 3339 and by the spelling it returns", async () => { + const admin = adminClient(); + const t = `ts_${testId().replace(/-/g, "_")}`; + + await chQuery( + `CREATE TABLE IF NOT EXISTS default.\`${t}\` (id String, at DateTime('UTC'), at3 DateTime64(3, 'UTC')) ENGINE = Memory`, + ); + await chQuery( + `INSERT INTO default.\`${t}\` VALUES ('r1', '2026-01-15 10:30:00', '2026-01-15 10:30:00.123'), ('r2', '2026-01-16 00:00:00', '2026-01-16 00:00:00.000')`, + ); + await admin.schema.refresh(); + + const currentPolicy = readPolicyFile(); + await setPolicy({ + tables: { + ...currentPolicy.tables, + [t]: { viewer: { select: { allow_columns: ["*"] } } }, + }, + }); + + try { + const ids = async (column: string, op: FilterOp, value: unknown) => { + const result = await wh + .from(t) + .select("id") + .where(column, op, value) + .orderBy("id", "asc") + .fetch(); + expect(result.error).toBeNull(); + return (result.data as { id: string }[]).map((r) => r.id); + }; + expect(await ids("at", "=", "2026-01-15T10:30:00Z")).toEqual(["r1"]); + expect(await ids("at", ">=", "2026-01-15T12:00:00+02:00")).toEqual(["r1", "r2"]); + expect(await ids("at3", "=", "2026-01-15T10:30:00.123Z")).toEqual(["r1"]); + expect(await ids("at", "in", ["2026-01-16T00:00:00Z"])).toEqual(["r2"]); + + const back = await wh.from(t).select("id", "at", "at3").where("id", "=", "r1").fetch(); + expect(back.error).toBeNull(); + const row = back.data![0] as { at: string; at3: string }; + expect(row.at).toBe("2026-01-15 10:30:00"); + expect(await ids("at", "=", row.at)).toEqual(["r1"]); + expect(await ids("at3", "=", row.at3)).toEqual(["r1"]); + } finally { + await setPolicy(currentPolicy); + await chQuery(`DROP TABLE IF EXISTS default.\`${t}\``); + } + }); + + it("refuses a null filter value with a 400", async () => { + const jwt = makeJWT({ sub: "test-viewer", role: "viewer", tenant_id: "acme" }); + const res = await fetch(`${WH_URL}/v1/query?table=${encodeURIComponent(T.clicks)}`, { + method: "POST", + headers: { "content-type": "application/json", authorization: `Bearer ${jwt}` }, + body: JSON.stringify({ + columns: ["event_id"], + filters: [{ column: "page", op: "eq", value: null }], + }), + }); + expect(res.status).toBe(400); + const body = (await res.json()) as { error: string }; + expect(body.error).toContain("must not be null"); + }); + it("rejects queries to unauthorized tables (403)", async () => { const admin = adminClient(); From 2aa6f7df8cd8cced0c18e3efd6fbf32107fce4ec Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 05:56:00 -0400 Subject: [PATCH 09/70] feat(config,ingest): chtypes registry key and typelayer insert settings clickhouse.chtypes_registry (WH_CHTYPES_REGISTRY) names the chtypes artifact directory the API role's type layer searches first; empty keeps the SDK's own search path. It is a bound key, so boot's refusal of unbound WH_* variables lets it through. The ingest worker takes its parsing settings from typelayer.InsertSettings, the map the API judges rows under, and pins async_insert=0 so a message is acked only once its row is stored. Nothing else in the worker changes; its test fixture builds rows by hand instead of through EncodeCompactRow. The dev and quickstart ClickHouse move to 26.8.15.10. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- config.yaml | 12 ++++-- deployments/compose/dependencies.yaml | 9 ++-- deployments/compose/standalone.yaml | 8 +++- internal/config/config.go | 21 ++++++---- internal/config/config_test.go | 12 ++++++ internal/ingest/worker.go | 29 ++++++------- internal/ingest/worker_test.go | 59 ++++++++++++++++++++------- internal/testutil/testutil.go | 4 +- 8 files changed, 108 insertions(+), 46 deletions(-) diff --git a/config.yaml b/config.yaml index e7f82ac4..9dbaf6c9 100644 --- a/config.yaml +++ b/config.yaml @@ -45,12 +45,18 @@ prometheus: path: /metrics port: 0 # 0 = mount on server.port; non-zero = sidecar listener -# Only the password and the connection ceiling are boot config; addr, -# http_port, http_scheme, database, username, query_timeout, tls, headers and -# the pool sizes are the settings directory's clickhouse block. +# Only the password, the connection ceiling and the chtypes artifact directory +# are boot config; addr, http_port, http_scheme, database, username, +# query_timeout, tls, headers and the pool sizes are the settings directory's +# clickhouse block. clickhouse: password: "" max_total_conns: 0 # ceiling on open native connections across pools; 0 = none + # chtypes artifact directory, read by api-role processes only. Empty (the + # default) defers to the SDK's search path: $CHTYPES_REGISTRY, the per-user + # cache ~/.cache/chtypes/artifacts/abi6/-, then the system + # directories. Set it only to point at a non-default location. + chtypes_registry: "" # Each layer's implementation, chosen at boot. Every layer defaults to its # in-process backend; mq and coord also take nats, cache redis, dedupe dynamodb. diff --git a/deployments/compose/dependencies.yaml b/deployments/compose/dependencies.yaml index 4cc4f2fd..77a25a73 100644 --- a/deployments/compose/dependencies.yaml +++ b/deployments/compose/dependencies.yaml @@ -11,10 +11,11 @@ name: wavehouse-dev services: clickhouse: - # Pinned to match tests/integration/setup_test.go (26.8 changed numeric - # DateTime64 parsing; see the comment there) so the dev server behaves - # like the one the tests assert against. - image: clickhouse/clickhouse-server:26.6.3.62 + # Pinned to match tests/integration/setup_test.go, so the dev server + # behaves like the one the tests assert against, and to a ClickHouse line + # chtypes.lock has an artifact for: the type layer answers with the + # artifact matching the server's own version. + image: clickhouse/clickhouse-server:26.8.15.10 ports: - "8123:8123" - "9000:9000" diff --git a/deployments/compose/standalone.yaml b/deployments/compose/standalone.yaml index d2fea9fe..bc3d1500 100644 --- a/deployments/compose/standalone.yaml +++ b/deployments/compose/standalone.yaml @@ -1,6 +1,7 @@ services: clickhouse: - image: clickhouse/clickhouse-server:26.6.3.62 + # A ClickHouse line chtypes.lock has an artifact for (the image bakes it). + image: clickhouse/clickhouse-server:26.8.15.10 ports: - "8123:8123" - "9000:9000" @@ -28,6 +29,11 @@ services: # ceiling are env; the bundled ClickHouse has no password. # WH_CH_PASSWORD: "" # WH_CH_MAX_TOTAL_CONNS: "0" + # deployments/Dockerfile bakes the pinned chtypes artifacts + # (chtypes.lock) into the image at /opt/chtypes/artifacts and sets + # CHTYPES_REGISTRY there, so the quickstart needs nothing here. To use a + # bind-mounted artifact directory instead of rebuilding the image: + # WH_CHTYPES_REGISTRY: /path/in/container # No auth on/off switch: set WH_AUTH_JWT_SECRET to validate tokens; # without it, every request resolves to the policy default_role. # WH_AUTH_JWT_SECRET: "change-me" diff --git a/internal/config/config.go b/internal/config/config.go index 99016ac9..01ca40d5 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -131,19 +131,26 @@ type Server struct { ShutdownTimeout int `yaml:"shutdown_timeout" env:"WH_SERVER_SHUTDOWN_TIMEOUT"` } -// ClickHouse holds the password and the connection ceiling. The wiring — -// address, HTTP port and scheme, database, username, query timeout, TLS, -// headers, pool size — is the settings directory's `clickhouse` block -// (hot-reloadable: a change swaps the connection). The password stays here -// because secrets don't belong in a tracked JSON file; it is combined with -// the adopted wiring on every (re)connect. The ceiling stays here because -// it is capacity, sized once per process, not wiring. +// ClickHouse holds the password, the connection ceiling and the chtypes +// artifact directory. The wiring — address, HTTP port and scheme, database, +// username, query timeout, TLS, headers, pool size — is the settings +// directory's `clickhouse` block (hot-reloadable: a change swaps the +// connection). The password stays here because secrets don't belong in a +// tracked JSON file; it is combined with the adopted wiring on every +// (re)connect. The ceiling and the artifact directory stay here because each +// is read once per process: capacity, and the registry the type layer opens +// at boot. type ClickHouse struct { Password string `yaml:"password" env:"WH_CH_PASSWORD"` // MaxTotalConns caps the native connections the process may hold open // across its pools: the settings directory's clickhouse.max_open_conns // must not exceed it. 0, the default, is no ceiling. MaxTotalConns int `yaml:"max_total_conns" env:"WH_CH_MAX_TOTAL_CONNS"` + // ChtypesRegistry is the chtypes artifact directory the type layer + // searches first (typelayer.Config.RegistryDir). Empty, the default, is + // the SDK's own search path: $CHTYPES_REGISTRY, the per-user cache, then + // the system directories. Only a process with the api role reads it. + ChtypesRegistry string `yaml:"chtypes_registry" env:"WH_CHTYPES_REGISTRY"` } // Auth holds the authentication secrets. The verifier wiring — `jwks_url`, diff --git a/internal/config/config_test.go b/internal/config/config_test.go index ee8b07cc..be06f8eb 100644 --- a/internal/config/config_test.go +++ b/internal/config/config_test.go @@ -40,6 +40,7 @@ func TestLoad_Defaults(t *testing.T) { assert.Equal(t, 10, cfg.Server.ShutdownTimeout) assert.Equal(t, "", cfg.ClickHouse.Password) assert.Equal(t, 0, cfg.ClickHouse.MaxTotalConns, "no connection ceiling by default") + assert.Empty(t, cfg.ClickHouse.ChtypesRegistry, "the SDK's own search path by default") assert.Empty(t, cfg.Auth.OperatorKey, "operator key is empty by default (feature off)") assert.Equal(t, "./data", cfg.DataDir) assert.False(t, cfg.OTel.Enabled) @@ -59,6 +60,7 @@ server: clickhouse: password: "ch-pass" max_total_conns: 40 + chtypes_registry: /opt/chtypes/artifacts auth: jwt_secret: "test-secret" operator_key: "op-key" @@ -72,6 +74,7 @@ auth: assert.Equal(t, 9090, cfg.Server.Port) assert.Equal(t, "ch-pass", cfg.ClickHouse.Password) assert.Equal(t, 40, cfg.ClickHouse.MaxTotalConns) + assert.Equal(t, "/opt/chtypes/artifacts", cfg.ClickHouse.ChtypesRegistry) assert.Equal(t, "test-secret", cfg.Auth.JWTSecret) assert.Equal(t, "op-key", cfg.Auth.OperatorKey) } @@ -83,6 +86,15 @@ func TestLoad_OperatorKey_FromEnv(t *testing.T) { assert.Equal(t, "env-operator-key", cfg.Auth.OperatorKey) } +// The variable is bound, so boot's refusal of unbound WH_* names lets it +// through and Load reads it. +func TestLoad_ChtypesRegistry_FromEnv(t *testing.T) { + t.Setenv("WH_CHTYPES_REGISTRY", "/srv/chtypes") + cfg, err := Load("nonexistent.yaml") + require.NoError(t, err) + assert.Equal(t, "/srv/chtypes", cfg.ClickHouse.ChtypesRegistry) +} + func TestLoad_EnvOverridesYAML(t *testing.T) { dir := t.TempDir() yamlContent := ` diff --git a/internal/ingest/worker.go b/internal/ingest/worker.go index 523eb1c8..8258a29b 100644 --- a/internal/ingest/worker.go +++ b/internal/ingest/worker.go @@ -20,6 +20,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/chsql" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" "go.opentelemetry.io/otel" "go.opentelemetry.io/otel/attribute" "go.opentelemetry.io/otel/metric" @@ -838,20 +839,20 @@ func (w *IngestWorker) insertToClickHouse(ctx context.Context, tableName string, q.Set("database", t.Database) q.Set("param_target_table", tableName) q.Set("query", fmt.Sprintf("INSERT INTO {target_table:Identifier} (%s) FORMAT JSONCompactEachRow", strings.Join(quoted, ", "))) - q.Set("date_time_input_format", "best_effort") - // A field the record omitted rides as an explicit null in its column's slot, - // because a positional row has one value per column and no way to say - // "absent". For a NON-nullable column with a default this setting turns that - // null back into the default, matching what omitting the key did under - // JSONEachRow. It is already the server default (verified on 26.6.3), so - // this is belt-and-braces for a server configured otherwise. - // - // TRANSITIONAL DIVERGENCE, and it is NOT what this setting controls: on a - // NULLABLE column an explicit null is stored as NULL whatever the setting - // says — only an ABSENT key ever took the default. So a `Nullable(T) DEFAULT - // …` column now stores NULL where it previously took its default. Verified - // on 26.6.3: omitted key → default; explicit null → NULL at both settings. - q.Set("input_format_null_as_default", "1") + // The parsing settings are the ones the API judged these rows under + // (typelayer), not literals of the worker's own: a setting the two did not + // share would ask the server a question chtypes never answered. Each row + // is ClickHouse's own export of its record, DEFAULTs already evaluated, so + // parsing it again under the same settings stores the values judged. + for k, v := range typelayer.InsertSettings() { + q.Set(k, v) + } + // Synchronous: the worker acks a message only once its row is in, so an + // async buffer would ack rows still in flight. Not a parsing setting, so + // chtypes never sees it. insert_deduplicate stays at the server default: + // idempotency is the gateway's (internal/dedupe), and a Replicated + // engine's block-hash dedupe is the operator's choice for their engine. + q.Set("async_insert", "0") req, err := http.NewRequestWithContext(ctx, "POST", t.URL+"?"+q.Encode(), &buf) if err != nil { diff --git a/internal/ingest/worker_test.go b/internal/ingest/worker_test.go index 26a6f747..09ab712d 100644 --- a/internal/ingest/worker_test.go +++ b/internal/ingest/worker_test.go @@ -30,10 +30,10 @@ import ( "github.com/Wave-RF/WaveHouse/internal/cache" "github.com/Wave-RF/WaveHouse/internal/chconn" - "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" + "github.com/Wave-RF/WaveHouse/internal/typelayer" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -70,24 +70,43 @@ func makeEnvelope(t *testing.T, tableName, scope string, data map[string]any) [] // that publish two different schemas for one table. func makeEnvelopeCols(t *testing.T, tableName, scope string, cols []string, data map[string]any) []byte { t.Helper() - schema := make([]discovery.Column, len(cols)) - for i, c := range cols { - schema[i] = discovery.Column{Name: c, Position: uint64(i + 1)} - } - row, err := EncodeCompactRow(schema, data) - require.NoError(t, err) out, err := json.Marshal(EventMessage{ TableName: tableName, Scope: scope, ReceivedTimestamp: "2026-01-01T00:00:00Z", Format: FormatJSONCompactEachRow, Columns: cols, - Row: row, + Row: compactRow(t, cols, data), }) require.NoError(t, err) return out } +// compactRow renders data as one JSONCompactEachRow row in cols order, a +// column the record omits encoding as null. Production rows are ClickHouse's +// own export (internal/typelayer) and the worker only forwards bytes, so a +// hand-built row is the right fixture here. +func compactRow(t *testing.T, cols []string, data map[string]any) json.RawMessage { + t.Helper() + var buf bytes.Buffer + buf.WriteByte('[') + for i, c := range cols { + if i > 0 { + buf.WriteByte(',') + } + v, ok := data[c] + if !ok { + buf.WriteString("null") + continue + } + b, err := json.Marshal(v) + require.NoError(t, err) + buf.Write(b) + } + buf.WriteByte(']') + return json.RawMessage(buf.Bytes()) +} + // newIngestMsg builds a MockMessage shaped exactly the way the // /v1/ingest producer (internal/api/ingest.go) publishes events: // @@ -328,13 +347,23 @@ func TestInsertToClickHouse_BuildsCorrectRequest(t *testing.T) { assert.Equal(t, "test_db", q.Get("database")) assert.Equal(t, "events", q.Get("param_target_table")) assert.Equal(t, "INSERT INTO {target_table:Identifier} (`id`) FORMAT JSONCompactEachRow", q.Get("query")) - // #372: the insert pins best_effort — the server default since - // ClickHouse 26.5; older 'basic' defaults reject the canonical - // form's zone suffix. - assert.Equal(t, "best_effort", q.Get("date_time_input_format")) - // A field the record omitted rides as null in its column's slot; - // this is what turns it back into the column's default. - assert.Equal(t, "1", q.Get("input_format_null_as_default")) + // The parsing settings are exactly the ones the API judged the + // rows under, plus a synchronous insert, and nothing else. + want := map[string]string{ + "database": "test_db", + "param_target_table": "events", + "query": q.Get("query"), + "async_insert": "0", + } + for k, v := range typelayer.InsertSettings() { + want[k] = v + } + got := map[string]string{} + for k := range q { + got[k] = q.Get(k) + } + assert.Equal(t, want, got) + assert.Equal(t, "best_effort", q.Get("date_time_input_format"), "#372: older 'basic' defaults reject a zone suffix") assert.Equal(t, "application/json", req.Header.Get("Content-Type")) assert.Equal(t, "test_user", req.Header.Get("X-ClickHouse-User")) diff --git a/internal/testutil/testutil.go b/internal/testutil/testutil.go index 0a8f0021..5f3b948f 100644 --- a/internal/testutil/testutil.go +++ b/internal/testutil/testutil.go @@ -22,8 +22,8 @@ import ( // NewTestSchemaRegistry creates a SchemaRegistry pre-loaded with the given // table schemas, without a real ClickHouse: a mock connection serves the // schemas as system.columns rows (UTC as the server zone) and the registry is -// built by the real discovery path — NewSchemaRegistry + Refresh — so -// timestamp column specs are precomputed exactly as in production. +// built by the real discovery path — NewSchemaRegistry + Refresh — so the +// derived fields are computed exactly as in production. // // The registry holds schemas rebuilt from those rows (Name, Type, HasDefault, // DefaultKind, DefaultExpression, Position, DDL; IsNullable derived from the From 852d5f81eac60f795209af22fb89b2d1872c0e1c Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:01:50 -0400 Subject: [PATCH 10/70] feat(stream): judge row-level security with the type layer, per tenant The hub's row filter is now decided by ClickHouse's own parser and expression engine through the tenant's compiled schema, instead of a Go re-implementation of ClickHouse's comparisons over a name-keyed decode. - RowEvaluator.Prepare(tenant, table, columns, row) parses each event once; the returned view answers per subscriber. NewRowEvaluator(engine) is the production evaluator; an unwired hub or a nil engine withholds every row of a row-filtered role instead of delivering it. - Rows published with the inserting role's narrower column list are parsed under that list and evaluated, in any column order. A filter on a column the event does not carry withholds the row for that subscriber, since the stored row's DEFAULT is computed again at insert. - wavehouse_sse_rows_withheld_total gains a reason label: filter, error, decline, unavailable or drift. - The policy read stays per event during replay; the schema frame, Prune and tenant topics are unchanged. - The SSE end-to-end test expects ClickHouse's DateTime64 rendering. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/stream/hub.go | 269 +++++++++++-------------- internal/stream/hub_test.go | 326 ++++++++++++++++++++---------- internal/stream/metrics.go | 21 +- internal/stream/roweval.go | 211 ++++++++++++++++++++ internal/stream/roweval_test.go | 338 ++++++++++++++++++++++++++++++-- tests/e2e/sdk/streaming.test.ts | 24 ++- 6 files changed, 907 insertions(+), 282 deletions(-) create mode 100644 internal/stream/roweval.go diff --git a/internal/stream/hub.go b/internal/stream/hub.go index b13221bf..6683e565 100644 --- a/internal/stream/hub.go +++ b/internal/stream/hub.go @@ -23,9 +23,10 @@ import ( // resolved against each subscriber's claims, so two subscribers of the same role can // be entitled to different rows. For a role that carries a row-filter, Broadcast // therefore keeps the shared column projection but evaluates row visibility PER -// subscriber (ResolvedPermissions.RowVisible) before delivering — closing the -// query/stream RLS drift in #319. Roles without a row-filter keep the pure -// once-per-role fast path unchanged. See projectColumns. +// subscriber (RowView.Visible over the event parsed once) before delivering — +// closing the query/stream RLS drift in #319. Roles without a row-filter keep the +// pure once-per-role fast path unchanged, and never reach the type layer at all. +// See projectIndices. // // A topic is one tenant's table (mq.Topic, #583): subscribers register under // the full topic, so one tenant's subscribers never see another's rows for a @@ -35,46 +36,35 @@ type Hub struct { mu sync.RWMutex topics map[mq.Topic]*topicRoutes policy PolicySource // nil ⇒ policy filtering not configured (legacy passthrough) - registry RegistrySource // nil, or yielding nil ⇒ no column types; row-filter comparison degrades fail-closed (see columnSpecs) + registry RegistrySource // nil, or yielding nil ⇒ no schema frame on subscribe; row filtering is unaffected (the type layer owns it) metric *Metrics // nil-safe - // RowEvaluator is the seam a native type layer will take over: the one - // place a row's visibility under a role's row-filter is decided. nil means - // the default implementation, which delegates to ResolvedPermissions.RowVisible - // — today's behavior unchanged. Wired once before the Hub serves traffic and - // not safe to mutate afterwards: rowAdmitted reads it from the consumer - // goroutines (one per tenant) and from SSE handler goroutines without - // holding h.mu. Every delivery path reaches it through rowAdmitted, never - // directly. + // RowEvaluator decides a row's visibility under a role's row-filter — the + // one place that decision is taken, for live fan-out and replay alike. + // Production wires NewRowEvaluator (the type layer); nil means the + // fail-closed default below, which withholds every row of a row-filtered + // role, because an unwired evaluator must never read as "everything + // visible". Wired once before the Hub serves traffic and not safe to mutate + // afterwards: it is read from the consumer goroutines (one per tenant) and + // from SSE handler goroutines without holding h.mu. Every delivery path + // reaches it through Prepare + rowAdmitted, never directly. RowEvaluator RowEvaluator -} - -// RowEvaluator decides whether one decoded event row is visible to a subscriber -// under their resolved permissions. specs classifies each column for the -// comparison (see Hub.columnSpecs); a nil map means no type knowledge, which the -// default implementation treats as the fail-closed floor. -type RowEvaluator interface { - Visible(perms *policy.ResolvedPermissions, row map[string]any, specs map[string]policy.ColumnSpec) bool -} - -// policyRowEvaluator is the default RowEvaluator, delegating to the policy -// package's in-memory row-filter evaluation. -type policyRowEvaluator struct{} -// Visible delegates to the policy package's in-memory row-filter evaluation, -// the same resolution the query path renders to SQL (#319). -func (policyRowEvaluator) Visible(perms *policy.ResolvedPermissions, row map[string]any, specs map[string]policy.ColumnSpec) bool { - return perms.RowVisible(row, specs) + // unwired is that fail-closed default: an engineEvaluator with no Engine, + // which withholds and logs once. It is a field rather than a fresh value per + // call so the "no engine" report really is once per Hub. + unwired engineEvaluator } -// rowEvaluator returns the Hub's RowEvaluator, or the default when none is -// wired. Nil-safe rather than constructor-enforced, so a zero Hub still -// evaluates row-level security instead of panicking past it. +// rowEvaluator returns the Hub's RowEvaluator, or the fail-closed default when +// none is wired. Nil-safe rather than constructor-enforced, so a zero Hub still +// enforces row-level security instead of panicking past it — or, worse, reading +// an absent evaluator as an absent restriction. func (h *Hub) rowEvaluator() RowEvaluator { if h.RowEvaluator != nil { return h.RowEvaluator } - return policyRowEvaluator{} + return &h.unwired } // topicRoutes holds the per-role buckets subscribed to one topic. @@ -94,17 +84,17 @@ type RegistrySource func(tenant.ID) *discovery.SchemaRegistry // NewHub builds the event hub. A nil policy store passes every event through // unfiltered (the unwired-tests case); a non-nil store whose Get returns nil is a -// total lockout (a deleted/absent policy denies everyone). A nil registry source leaves -// every column's type unknown, so row-filter comparison degrades FAIL-CLOSED: -// equality/set predicates admit only a byte-identical value and ordering/!= admit -// nothing (see policy.ColumnKind); metric may be nil. +// total lockout (a deleted/absent policy denies everyone). A nil registry source +// only costs the on-subscribe schema frame — row filtering reads its types from +// the type layer behind RowEvaluator, which the caller wires separately; metric +// may be nil. func NewHub(policyStore PolicySource, registry RegistrySource, metric *Metrics) *Hub { return &Hub{topics: make(map[mq.Topic]*topicRoutes), policy: policyStore, registry: registry, metric: metric} } // schema is tenant id's schema for table, nil when the hub has no registry // source, the tenant no registry, or the registry no such table — every one -// of them the fail-closed reading. +// of them read as nothing to announce. func (h *Hub) schema(id tenant.ID, table string) *discovery.TableSchema { if h.registry == nil { return nil @@ -226,11 +216,18 @@ func (h *Hub) Broadcast(topic mq.Topic, raw []byte) { ev := newEventView(raw) p, filter := h.snapshotPolicy(topic.Tenant) - // Column specs for type-aware row-filter comparison — resolved lazily at most - // once per event, only when some role actually carries a row-filter, and reused - // across every filtered role and subscriber. - var colSpecs map[string]policy.ColumnSpec - specsResolved := false + // The row is parsed at most ONCE per event — only when some role actually + // carries a row-filter — and the parsed view is reused by every filtered role + // and subscriber. Parsing is the expensive half; evaluating a compiled + // predicate over an already-parsed row is not. + var view RowView + var prepErr error + prepared := false + defer func() { + if view != nil { + view.Close() + } + }() for _, rb := range roleBuckets { plan, ok := planForRole(p, filter, rb.role, ev, KindEvent) @@ -255,12 +252,12 @@ func (h *Hub) Broadcast(topic mq.Topic, raw []byte) { // shared, but whether each subscriber may see THIS row depends on its claims, so // evaluate visibility per subscriber. Predicates read the full event row (a // filter may key on a column the role can't SELECT), not the projected columns. - if !specsResolved { - colSpecs = h.columnSpecs(topic.Tenant, ev.evt.TableName) - specsResolved = true + if !prepared { + view, prepErr = h.rowEvaluator().Prepare(topic.Tenant, ev.evt.TableName, ev.evt.Columns, ev.evt.Row) + prepared = true } for _, sub := range rb.bucket.Snapshot() { - if h.rowAdmitted(p, rb.role, ev, sub.claims, colSpecs) { + if h.rowAdmitted(p, rb.role, ev, sub.claims, view, prepErr) { deliver(sub, plan) } } @@ -293,62 +290,34 @@ func deliver(sub *Subscriber, plan rolePlan) { } // rowAdmitted reports whether claims admit this event's row under the role's -// row-filter, counting a withheld row when they don't. It is the one admission -// step shared by the live fan-out (per subscriber) and replay (per connection), -// so the two delivery paths can't drift on how row-level security is evaluated. -func (h *Hub) rowAdmitted(p *policy.Policy, role string, ev *eventView, claims map[string]any, colSpecs map[string]policy.ColumnSpec) bool { +// row-filter, counting a withheld row (with the reason) when they don't. It is +// the one admission step shared by the live fan-out (per subscriber) and replay +// (per connection), so the two delivery paths can't drift on how row-level +// security is evaluated. +// +// view is the event parsed once for the whole fan-out; prepErr is why there is +// none. An event that could not be prepared — no compiled schema for the +// tenant's table, a column list the table cannot have — withholds from every +// row-filtered subscriber, never from the unfiltered roles that never asked the +// type layer anything. +func (h *Hub) rowAdmitted(p *policy.Policy, role string, ev *eventView, claims map[string]any, view RowView, prepErr error) bool { + if prepErr != nil || view == nil { + // view == nil with no error is a broken evaluator, not a verdict: withhold. + h.metric.RowWithheld(ev.evt.TableName, role, WithheldReason(prepErr)) + return false + } perms := policy.Evaluate(p, role, ev.evt.TableName, "select", claims) - if !h.rowEvaluator().Visible(perms, ev.row, colSpecs) { - h.metric.RowWithheld(ev.evt.TableName, role) + if visible, reason := view.Visible(perms); !visible { + h.metric.RowWithheld(ev.evt.TableName, role, reason) return false } return true } -// columnSpecs classifies each of the table's columns for the row-filter evaluator: -// DateTime/DateTime64 columns compare as instants (through discovery's -// Column.TimeParser — the same grammar ingest canonicalization applies, so a -// zone-less filter constant matches the canonical RFC 3339 payload), numeric types -// compare numerically (9 < 100, matching ClickHouse), String compares bytewise -// (exactly ClickHouse's String collation), and any other type is omitted — -// policy.ColumnOpaque, the map's zero value — admitting byte-equality only. nil when -// no schema is available (unknown table, or a Hub built without a registry), which -// reads as every column Opaque: the fail-closed floor, never a lexicographic -// fallback that could admit rows the query path excludes ("9" > "100" as text). -func (h *Hub) columnSpecs(id tenant.ID, table string) map[string]policy.ColumnSpec { - schema := h.schema(id, table) - if schema == nil { - return nil - } - m := make(map[string]policy.ColumnSpec, len(schema.Columns)) - for _, c := range schema.Columns { - if pt := c.TimeParser(); pt != nil { - m[c.Name] = policy.ColumnSpec{Kind: policy.ColumnTime, ParseTime: pt} - continue - } - switch { - case discovery.IsNumericType(c.Type): - // The storage model narrows both comparison operands the way - // ClickHouse narrows the stored value and the bound constant. A - // numeric type whose model can't be classified keeps the zero - // NumericSpec, which refuses every comparison — fail closed, - // never a comparison under guessed semantics. - spec := policy.ColumnSpec{Kind: policy.ColumnNumeric} - if st, ok := discovery.NumericStorageOf(c.Type); ok { - spec.Numeric = NumericSpecOf(st) - } - m[c.Name] = spec - case discovery.IsStringType(c.Type): - m[c.Name] = policy.ColumnSpec{Kind: policy.ColumnText} - } - } - return m -} - -// NumericSpecOf renders discovery's storage classification as the policy -// evaluator's storage model. Exported so the tests/integration differential -// oracle builds specs through the very mapping production uses — one source, -// so the oracle can't keep validating a mapping the Hub no longer applies. +// NumericSpecOf renders discovery's storage classification as the Go row-filter +// evaluator's storage model. The Hub no longer uses it — row filtering is the +// type layer's — and it is kept only for the tests/integration oracle that +// still calls it, until that test and the Go evaluator are removed together. func NumericSpecOf(st discovery.NumericStorage) policy.NumericSpec { switch { case st.Integer: @@ -360,11 +329,11 @@ func NumericSpecOf(st discovery.NumericStorage) policy.NumericSpec { } } -// eventView is one published event decoded once per Broadcast, in the two forms -// the delivery paths need: cells, the raw JSON value at each envelope column -// position (sliced positionally into the outgoing frame, so a value's bytes are -// never re-encoded), and row, the same values keyed by column name for the -// row-filter evaluator. +// eventView is one published event decoded once per Broadcast: cells, the raw +// JSON value at each envelope column position, sliced positionally into the +// outgoing frame so a value's bytes are never re-encoded. The row-filter reads +// the ORIGINAL positional bytes (ev.evt.Row) through the type layer rather than +// any decoding done here — the parse that decides visibility is ClickHouse's own. // // raw and decoded carry the legacy no-policy passthrough: a payload that is not // an EventMessage at all is forwarded verbatim when no policy store is wired, @@ -375,16 +344,13 @@ type eventView struct { decoded bool // raw parsed as an EventMessage usable bool // ...and it declares a known format whose columns and row pair cells []json.RawMessage - row map[string]any } -// newEventView decodes raw once for the whole fan-out. Numbers decode as -// json.Number — exact digit strings, not float64 — because the row-filter -// comparison must see the same value ClickHouse stores: ingest decodes with -// UseNumber and forwards the row verbatim, so a bare 64-bit ID past 2^53 keeps -// its exact digits on the query path, and a lossy float64 decode here would -// collapse neighboring IDs into one value and deliver another tenant's row. The -// outgoing frame reuses the raw cell bytes, so it stays byte-faithful regardless. +// newEventView decodes raw once for the whole fan-out. The envelope's own fields +// decode with UseNumber so nothing in it is rounded, and the row's cells are kept +// as raw bytes: the outgoing frame reuses them verbatim, so a 64-bit id past 2^53 +// keeps every digit on the wire, and the visibility decision never sees a Go +// float at all. func newEventView(raw []byte) *eventView { ev := &eventView{raw: raw} if !decodeEvent(raw, &ev.evt) { @@ -399,23 +365,26 @@ func newEventView(raw []byte) *eventView { if ev.evt.Format != ingest.FormatJSONCompactEachRow { return ev } - ev.cells, ev.row, ev.usable = pairRow(ev.evt.Columns, ev.evt.Row) + ev.cells, ev.usable = pairRow(ev.evt.Columns, ev.evt.Row) return ev } -// pairRow splits a compact row into its cells and zips them with the column -// names. ok is false when the two cannot be paired — an undecodable row, a +// pairRow splits a compact row into its cells and checks that they pair with the +// column names. ok is false when the two cannot be paired — an undecodable row, a // length that disagrees with the column list, a repeated column name, or an // empty column list (which no length check catches, since a zero-length row // agrees with it) — because there is then no way to say which value belongs to -// which column, and a row-filter that cannot read its column must withhold -// rather than guess. -func pairRow(cols []string, row json.RawMessage) (cells []json.RawMessage, byName map[string]any, ok bool) { +// which column, and neither the announced schema frame nor a row-filter may +// guess. +// +// The values themselves are never decoded here: they go to the type layer as +// the bytes they arrived as, and out to the client as the same bytes. +func pairRow(cols []string, row json.RawMessage) (cells []json.RawMessage, ok bool) { if len(row) == 0 { - return nil, nil, false + return nil, false } if err := json.Unmarshal(row, &cells); err != nil { - return nil, nil, false + return nil, false } // A zero-column envelope pairs with anything of length zero — both `null`, // which unmarshals to a nil slice, and `[]` — and would then be announced as @@ -424,32 +393,26 @@ func pairRow(cols []string, row json.RawMessage) (cells []json.RawMessage, byNam // (parseMsg's len(envelope.Columns) == 0), so refuse it here too rather than // let the two consumers disagree about an envelope neither can read. if len(cols) == 0 { - return nil, nil, false + return nil, false } if len(cells) != len(cols) { - return nil, nil, false + return nil, false } - byName = make(map[string]any, len(cols)) - for i, c := range cols { - // A repeated name has no single meaning: the map would keep the last - // value and silently drop the first, and this map is what a row-level - // filter is evaluated against — so a duplicate could decide visibility - // on a value the row never really carried. Unpairable, like a length - // mismatch. Cannot arise from our own producer (the envelope's columns - // come from system.columns, where ClickHouse forbids two columns of one - // name), so this is defence for an envelope we did not write. - if _, dup := byName[c]; dup { - return nil, nil, false - } - dec := json.NewDecoder(bytes.NewReader(cells[i])) - dec.UseNumber() - var v any - if err := dec.Decode(&v); err != nil { - return nil, nil, false + // A repeated name has no single meaning: the client zips the announced list + // against the positional row, so a duplicate makes two positions + // indistinguishable to it, and the worker's INSERT naming it is refused + // (code 15). Unpairable, like a length mismatch. Cannot arise from our own + // producer (the envelope's columns come from the table's own column list, + // where ClickHouse forbids two columns of one name), so this is defence for + // an envelope we did not write. + seen := make(map[string]struct{}, len(cols)) + for _, c := range cols { + if _, dup := seen[c]; dup { + return nil, false } - byName[c] = v + seen[c] = struct{}{} } - return cells, byName, true + return cells, true } // decodeEvent parses raw as a published EventMessage, reporting whether it is one @@ -485,19 +448,15 @@ func (h *Hub) snapshotPolicy(id tenant.ID) (p *policy.Policy, filter bool) { // projects a single replayed event for the connection's role+claims into a // ready-to-write replay frame, or ok=false to skip it (denied table, invalid // payload, or a row the claims aren't entitled to see). It is a Hub method so -// replay shares the Hub's policy store and schema registry with the live fan-out — +// replay shares the Hub's policy store and row evaluator with the live fan-out — // the handler can't accidentally project replay against a different (or nil) // policy. Replay is already per-connection, so row-level security evaluates against // this connection's claims directly; the returned closure reads the policy per // replayed event (matching Broadcast, so a reload landing mid-replay applies to -// the next replayed row) and caches only the per-table column-kind lookup across -// the replay loop, so a large Last-Event-ID gap-fill doesn't pay a registry -// lookup and map build per event. +// the next replayed row). // The closure is for a single goroutine — each connection makes its own. The live // path uses Broadcast. func (h *Hub) ReplayProjector(id tenant.ID, role string, sub *Subscriber) func(raw []byte) []Frame { - var colSpecs map[string]policy.ColumnSpec - specsFor := "" // table name colSpecs was resolved for ("" ⇒ not yet resolved) // Schema-drift state is LOCAL to this gap-fill, not the connection's shared // lastSchema. Replay writes straight to the socket while live events queue // behind it, so sharing the state would let a live event's announcement @@ -529,13 +488,15 @@ func (h *Hub) ReplayProjector(id tenant.ID, role string, sub *Subscriber) func(r return nil } if plan.perms.HasRowFilter() { - // One topic ⇒ one table, so this resolves once per replay in practice; the - // guard re-resolves if a stream ever mixes tables rather than going stale. - if specsFor != ev.evt.TableName { - colSpecs = h.columnSpecs(id, ev.evt.TableName) - specsFor = ev.evt.TableName + // Replay is per connection, so one prepared row serves exactly one + // visibility question — and it is closed immediately, because the + // gap-fill loop can run for thousands of events. + view, err := h.rowEvaluator().Prepare(id, ev.evt.TableName, ev.evt.Columns, ev.evt.Row) + admitted := h.rowAdmitted(p, role, ev, sub.claims, view, err) + if view != nil { + view.Close() } - if !h.rowAdmitted(p, role, ev, sub.claims, colSpecs) { + if !admitted { return nil // this row is filtered out for these claims } } @@ -575,9 +536,11 @@ func (h *Hub) SubscribeSchemaFrame(id tenant.ID, table, role string, sub *Subscr return Frame{}, false } } - // The insertable subset, matching what an event's envelope carries — a + // The insertable subset, matching what a full-width envelope carries — a // computed column never appears in a published row, so announcing it here - // would guarantee a drift re-announcement on the very first event. + // would guarantee a drift re-announcement on the very first event. An + // envelope from a role that may write fewer columns carries fewer, and is + // announced again when it arrives, like any other change of list. _, projected := projectIndices(schema.InsertableColumnNames(), perms) sig := schemaSignature(table, projected) if !sub.needsSchema(sig) { diff --git a/internal/stream/hub_test.go b/internal/stream/hub_test.go index 4510b475..f6d3b715 100644 --- a/internal/stream/hub_test.go +++ b/internal/stream/hub_test.go @@ -1,6 +1,7 @@ package stream import ( + "bytes" "context" "encoding/json" "fmt" @@ -21,6 +22,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" + "github.com/Wave-RF/WaveHouse/internal/typelayer" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" "go.opentelemetry.io/otel" @@ -63,11 +65,10 @@ func rawEvent(tb testing.TB, table, ts string, data map[string]any) []byte { } // TestBroadcast_DuplicateColumnWithheld: an envelope whose columns name one -// column twice cannot be paired — the name-keyed map would keep the last value -// and silently drop the first, and that map is what a row-level filter is -// evaluated against, so a duplicate could decide visibility on a value the row -// never carried. Withheld from every role rather than delivered under a guessed -// reading, the same as a length mismatch. +// column twice cannot be paired — the client zips the announced list against +// the positional row, so two positions under one name are indistinguishable to +// it, and no INSERT may name a column twice. Withheld from every role rather +// than delivered under a guessed reading, the same as a length mismatch. // // A nil policy store (passthrough, no filtering) and a positive control are // both load-bearing: with a restrictive policy the row would be withheld for @@ -170,10 +171,12 @@ func TestEventView_UnknownFormatWithheld(t *testing.T) { } // TestPairRow_Verdict enumerates the shapes that cannot be paired. Each one has -// to fail closed: a row-filter is evaluated against the name-keyed map, so a -// value read under the wrong name decides visibility on data the row never -// carried. The pairable case is the control that keeps the other seven honest — -// a pairRow that rejected everything would satisfy them alone. +// to fail closed: the announced column list is what the client zips the +// positional row against, and it is also the list the type layer reads the row +// under, so a shape where the two cannot be lined up would decide visibility — +// and label values — on data the row never carried. The pairable case is the +// control that keeps the other seven honest: a pairRow that rejected everything +// would satisfy them alone. func TestPairRow_Verdict(t *testing.T) { t.Parallel() for _, tt := range []struct { @@ -197,11 +200,10 @@ func TestPairRow_Verdict(t *testing.T) { } { t.Run(tt.name, func(t *testing.T) { t.Parallel() - cells, byName, ok := pairRow(tt.cols, json.RawMessage(tt.row)) + cells, ok := pairRow(tt.cols, json.RawMessage(tt.row)) assert.Equal(t, tt.ok, ok) if !tt.ok { assert.Nil(t, cells, "an unpairable envelope yields no cells to splice into a frame") - assert.Nil(t, byName, "and no map for a row-filter to read") } }) } @@ -211,12 +213,26 @@ func TestPairRow_Verdict(t *testing.T) { // declaration order or publish a column the record omits. func rawEventCols(tb testing.TB, table, ts string, cols []string, data map[string]any) []byte { tb.Helper() - schema := make([]discovery.Column, len(cols)) + // The positional line the ingest path publishes: one cell per column, in + // order, a column the record omits as null. Built here rather than through a + // production encoder because the wire shape is what these tests pin. + var buf bytes.Buffer + buf.WriteByte('[') for i, c := range cols { - schema[i] = discovery.Column{Name: c, Position: uint64(i + 1)} + if i > 0 { + buf.WriteByte(',') + } + v, ok := data[c] + if !ok { + buf.WriteString("null") + continue + } + b, err := json.Marshal(v) + require.NoError(tb, err) + buf.Write(b) } - row, err := ingest.EncodeCompactRow(schema, data) - require.NoError(tb, err) + buf.WriteByte(']') + row := json.RawMessage(buf.Bytes()) raw, err := json.Marshal(ingest.EventMessage{ TableName: table, ReceivedTimestamp: ts, @@ -459,6 +475,40 @@ func TestHub_ProjectsPerRole_DistinctRolesGetDistinctFrames(t *testing.T) { assert.NotEqual(t, fv.Data, fe.Data, "distinct role projections produce distinct bytes") } +// chtypesTable declares a table the way discovery would, numbering the columns +// in the order given. +func chtypesTable(name string, cols ...discovery.Column) *discovery.TableSchema { + for i := range cols { + cols[i].Position = uint64(i + 1) + } + return &discovery.TableSchema{Name: name, Columns: cols} +} + +// col is chtypesTable's shorthand for an ordinary stored column. +func col(name, chType string) discovery.Column { + return discovery.Column{Name: name, Type: chType} +} + +// chtypesHub is a Hub whose row filtering is decided the way production decides +// it: by the type layer, against the real ClickHouse 26.6 artifact, with the +// tables bound for the default tenant. Every row-filter test goes through this +// rather than a stub, because the verdicts under test ARE ClickHouse's — +// storage-domain narrowing, instant equality across spellings, exactness past +// 2^53 — and a stub could only restate what the test already believes. +func chtypesHub(tb testing.TB, store PolicySource, metric *Metrics, tables ...*discovery.TableSchema) *Hub { + tb.Helper() + hub := NewHub(store, nil, metric) + hub.RowEvaluator = NewRowEvaluator(typelayer.TestEngine(tb, tables...)) + return hub +} + +// clicksTable is the table the row-filter tests publish into: the readable +// column, one the viewer role may not select, and the filtered one. +// Declaration order matches the order rawEvent publishes (sorted by name). +func clicksTable() *discovery.TableSchema { + return chtypesTable("clicks", col("page", "String"), col("secret", "String"), col("tenant_id", "String")) +} + // rowFilterPolicy scopes role "viewer" to column "page" only, and to rows whose // tenant_id equals the caller's {{ jwt.tenant }} claim. The filter keys on tenant_id // — a column viewer may NOT select — so it also exercises the rule that row @@ -484,7 +534,7 @@ func rowFilterPolicy() *policy.Policy { // matching the constant-false predicate the query path binds for it. func TestHub_RowFilter_PerSubscriberIsolation(t *testing.T) { t.Parallel() - hub := NewHub(staticPolicy(rowFilterPolicy()), nil, nil) + hub := chtypesHub(t, staticPolicy(rowFilterPolicy()), nil, clicksTable()) topic := topicOf("clicks") acme := NewSubscriber(jwtClaims(t, map[string]any{"tenant": "acme"}), nil) @@ -506,7 +556,9 @@ func TestHub_RowFilter_PerSubscriberIsolation(t *testing.T) { // Unresolvable claim ⇒ no rows on the stream, matching the query path (#457). assertNoFrame(t, noTenant) - // A globex row reaches only the globex subscriber. + // A globex row reaches only the globex subscriber. Its envelope carries no + // "secret" — the shape a role that may not write that column publishes — + // and is evaluated all the same. hub.Broadcast(topic, rawEvent(t, "clicks", "2026-06-26T00:00:01Z", map[string]any{"tenant_id": "globex", "page": "/g"})) _, _, grow := recvEvent(t, globex) @@ -528,7 +580,8 @@ func TestHub_RowFilter_ClaimsSnapshotImmuneToCallerMutation(t *testing.T) { "clicks": {"viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"tenant_id": {Eq: new("{{ jwt.org.tenant }}")}}}}}, }, } - hub := NewHub(staticPolicy(p), nil, nil) + tenantTable := chtypesTable("clicks", col("page", "String"), col("tenant_id", "String")) + hub := chtypesHub(t, staticPolicy(p), nil, tenantTable) topic := topicOf("clicks") org := map[string]any{"tenant": "globex"} @@ -554,7 +607,7 @@ func TestHub_RowFilter_ClaimsSnapshotImmuneToCallerMutation(t *testing.T) { "clicks": {"viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"tenant_id": {In: new("{{ jwt.tenants }}")}}}}}, }, } - inHub := NewHub(staticPolicy(inPolicy), nil, nil) + inHub := chtypesHub(t, staticPolicy(inPolicy), nil, tenantTable) tenants := []any{"globex"} inSub := NewSubscriber(map[string]any{"tenants": tenants}, nil) inHub.Add(topic, "viewer", inSub) @@ -571,10 +624,12 @@ func TestHub_RowFilter_ClaimsSnapshotImmuneToCallerMutation(t *testing.T) { } // TestHub_RowFilter_MissingColumn_FailsClosed: an event that lacks the filtered -// column can't be proven visible, so it is withheld rather than leaked. +// column can't be proven visible, so it is withheld rather than leaked. The +// control proves the withholding is the absent column's doing, not a silent +// hub: the same subscriber gets the next event, which carries it. func TestHub_RowFilter_MissingColumn_FailsClosed(t *testing.T) { t.Parallel() - hub := NewHub(staticPolicy(rowFilterPolicy()), nil, nil) + hub := chtypesHub(t, staticPolicy(rowFilterPolicy()), nil, clicksTable()) topic := topicOf("clicks") acme := NewSubscriber(map[string]any{"tenant": "acme"}, nil) @@ -583,6 +638,11 @@ func TestHub_RowFilter_MissingColumn_FailsClosed(t *testing.T) { hub.Broadcast(topic, rawEvent(t, "clicks", "2026-06-26T00:00:00Z", map[string]any{"page": "/a"})) // no tenant_id assertNoFrame(t, acme) + + hub.Broadcast(topic, rawEvent(t, "clicks", "2026-06-26T00:00:01Z", + map[string]any{"page": "/b", "tenant_id": "acme"})) + _, _, row := recvEvent(t, acme) + assert.Equal(t, "/b", row["page"]) } // TestHub_RowFilter_SharedProjectionAcrossSameClaims: the column projection is still @@ -591,7 +651,7 @@ func TestHub_RowFilter_MissingColumn_FailsClosed(t *testing.T) { // per-subscriber, not the serialization. func TestHub_RowFilter_SharedProjectionAcrossSameClaims(t *testing.T) { t.Parallel() - hub := NewHub(staticPolicy(rowFilterPolicy()), nil, nil) + hub := chtypesHub(t, staticPolicy(rowFilterPolicy()), nil, clicksTable()) topic := topicOf("clicks") a := NewSubscriber(map[string]any{"tenant": "acme"}, nil) @@ -609,27 +669,21 @@ func TestHub_RowFilter_SharedProjectionAcrossSameClaims(t *testing.T) { assert.Same(t, &fa.Data[0], &fb.Data[0], "one serialization shared across same-role subscribers") } -// TestHub_RowFilter_NumericOrdering_SchemaInformed drives the registry-backed path: -// with a numeric column type in the schema, an `amount > 100` filter compares -// numerically, so amount=9 is withheld (a lexicographic "9" > "100" comparison would -// have leaked it) and amount=250 is delivered. Without a registry the same ordering -// filter has no type to trust and withholds every row — fail closed, never the -// lexicographic leak (the schemaless window is real: boot-time discovery failure -// retries in the background while the server serves). -func TestHub_RowFilter_NumericOrdering_SchemaInformed(t *testing.T) { +// TestHub_RowFilter_NumericOrdering drives the type-layer path: on a UInt64 +// column an `amount > 100` filter compares the way ClickHouse compares, so +// amount=9 is withheld (a lexicographic "9" > "100" would have leaked it) and +// amount=250 is delivered. With no type layer wired there is nothing that can +// answer the question at all, so every row is withheld — fail closed, never the +// lexicographic leak. +func TestHub_RowFilter_NumericOrdering(t *testing.T) { t.Parallel() - reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ - {Name: "clicks", Columns: []discovery.Column{ - {Name: "amount", Type: "UInt64"}, - {Name: "page", Type: "String"}, - }}, - }) p := &policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": {"viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"amount": {Gt: new("100")}}}}}, }, } - hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) + hub := chtypesHub(t, staticPolicy(p), nil, + chtypesTable("clicks", col("amount", "UInt64"), col("page", "String"))) topic := topicOf("clicks") sub := NewSubscriber(nil, nil) // constant filter value ⇒ no claims needed @@ -642,34 +696,31 @@ func TestHub_RowFilter_NumericOrdering_SchemaInformed(t *testing.T) { _, _, row := recvEvent(t, sub) assert.Equal(t, float64(250), row["amount"]) - // Same policy, no schema registry: an ordering predicate can't be proven either - // way, so both rows are withheld — including the one the schema-informed path - // delivers above. - noSchema := NewHub(staticPolicy(p), nil, nil) + // Same policy, no type layer: an ordering predicate can't be answered either + // way, so both rows are withheld — including the one the wired path delivers. + noEngine := NewHub(staticPolicy(p), nil, nil) blind := NewSubscriber(nil, nil) - noSchema.Add(topic, "viewer", blind) - noSchema.Broadcast(topic, rawEvent(t, "clicks", "t1", map[string]any{"amount": float64(9), "page": "/a"})) - noSchema.Broadcast(topic, rawEvent(t, "clicks", "t2", map[string]any{"amount": float64(250), "page": "/b"})) + noEngine.Add(topic, "viewer", blind) + noEngine.Broadcast(topic, rawEvent(t, "clicks", "t1", map[string]any{"amount": float64(9), "page": "/a"})) + noEngine.Broadcast(topic, rawEvent(t, "clicks", "t2", map[string]any{"amount": float64(250), "page": "/b"})) assertNoFrame(t, blind) } -// TestHub_RowFilter_FloatNarrowing_SchemaInformed drives storage-domain -// narrowing end-to-end through the registry: on a Float32 column, payload -// 16777217 stores as 16777216, so a `_gt: "16777216"` filter must withhold the -// event — the query path's WHERE over the stored row is false, and delivering -// the pre-narrowing payload was the ordering fail-open raised in review. A -// Float32-representable greater value still delivers. -func TestHub_RowFilter_FloatNarrowing_SchemaInformed(t *testing.T) { +// TestHub_RowFilter_FloatNarrowing drives storage-domain narrowing end-to-end: +// on a Float32 column, payload 16777217 stores as 16777216, so a +// `_gt: "16777216"` filter must withhold the event — the query path's WHERE over +// the stored row is false, and delivering the pre-narrowing payload was the +// ordering fail-open raised in review. A Float32-representable greater value +// still delivers. The row is parsed into the column's real storage before +// anything is compared. +func TestHub_RowFilter_FloatNarrowing(t *testing.T) { t.Parallel() - reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ - {Name: "clicks", Columns: []discovery.Column{{Name: "score", Type: "Float32"}}}, - }) p := &policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": {"viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"score": {Gt: new("16777216")}}}}}, }, } - hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) + hub := chtypesHub(t, staticPolicy(p), nil, chtypesTable("clicks", col("score", "Float32"))) topic := topicOf("clicks") sub := NewSubscriber(nil, nil) hub.Add(topic, "viewer", sub) @@ -682,6 +733,43 @@ func TestHub_RowFilter_FloatNarrowing_SchemaInformed(t *testing.T) { assert.Equal(t, float64(16777218), row["score"]) } +// TestHub_RowFilter_Float32EqualityBindsInTheColumnsWidth: the constant has to +// be read in the COLUMN's float domain, not the widest one. A Float32 column +// stores 0.1 as 0.100000001490116…, which is not Float64's 0.1 — so binding the +// constant as Float64 would make `= "0.1"` withhold the row and `!= "0.1"` +// ADMIT it, on a row /v1/query returns. +func TestHub_RowFilter_Float32EqualityBindsInTheColumnsWidth(t *testing.T) { + t.Parallel() + for _, tt := range []struct { + name string + filter policy.Filter + delivered bool + }{ + {"equality matches the stored Float32", policy.Filter{Eq: new("0.1")}, true}, + {"inequality does not", policy.Filter{Neq: new("0.1")}, false}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + p := &policy.Policy{Tables: map[string]policy.TablePolicy{ + "clicks": {"viewer": {Select: &policy.SelectPermissions{ + Filter: map[string]policy.Filter{"score": tt.filter}, + }}}, + }} + hub := chtypesHub(t, staticPolicy(p), nil, chtypesTable("clicks", col("score", "Float32"))) + sub := NewSubscriber(nil, nil) + hub.Add(topicOf("clicks"), "viewer", sub) + + hub.Broadcast(topicOf("clicks"), rawEvent(t, "clicks", "t", map[string]any{"score": json.Number("0.1")})) + if tt.delivered { + f, _, _ := recvEvent(t, sub) + assert.NotEmpty(t, f.Data) + } else { + assertNoFrame(t, sub) + } + }) + } +} + func TestHub_TopicIsolation(t *testing.T) { t.Parallel() hub := NewHub(nil, nil, nil) @@ -891,7 +979,7 @@ func TestHub_ReplayProjector_ReadsPolicyPerEvent(t *testing.T) { // gap-fill event is projected only when the connection's claims satisfy the filter. func TestHub_ReplayProjector_RowFilter(t *testing.T) { t.Parallel() - hub := NewHub(staticPolicy(rowFilterPolicy()), nil, nil) + hub := chtypesHub(t, staticPolicy(rowFilterPolicy()), nil, clicksTable()) raw := rawEvent(t, "clicks", "2026-06-26T00:00:00Z", map[string]any{"tenant_id": "acme", "page": "/a", "secret": "x"}) @@ -907,8 +995,8 @@ func TestHub_ReplayProjector_RowFilter(t *testing.T) { assert.NotContains(t, row, "secret", "denied column stripped on replay too") // The projector is reusable across a replay loop: a second event through the - // same closure (cached column kinds) projects identically — and does NOT - // re-announce a column list the connection already has. + // same closure projects identically — and does NOT re-announce a column + // list the connection already has. again := project(raw) require.Len(t, again, 1, "the column list is announced once per connection") assert.Equal(t, frames[1].Data, again[0].Data) @@ -954,9 +1042,9 @@ func TestHub_ConcurrentAddRemoveBroadcast_Race(t *testing.T) { // racing silently on a security decision. func TestHub_ConcurrentRowFilteredBroadcast_Race(t *testing.T) { t.Parallel() - hub := NewHub(staticPolicy(rowFilterPolicy()), nil, nil) + hub := chtypesHub(t, staticPolicy(rowFilterPolicy()), nil, clicksTable()) topic := topicOf("clicks") - raw := rawEvent(t, "clicks", "t", map[string]any{"tenant_id": "acme", "page": "/a"}) + raw := rawEvent(t, "clicks", "t", map[string]any{"tenant_id": "acme", "page": "/a", "secret": "x"}) var wg sync.WaitGroup for range 4 { // broadcasters: per-subscriber claims evaluation on every event @@ -984,26 +1072,23 @@ func TestHub_ConcurrentRowFilteredBroadcast_Race(t *testing.T) { } // TestHub_RowFilter_BigIntegerExact: a bare JSON integer past 2^53 must keep its -// exact digits through the hub's decode (UseNumber), or the row filter compares a +// exact digits all the way to the comparison, or the row filter compares a // lossily-rounded value: tenant 10000000000000001's row would falsely equal a // tenant claim of 10000000000000000 — float64 collapses the neighbors — and be // delivered cross-tenant on the stream while the query path (ClickHouse stores the -// exact digits ingest forwarded) excludes it. The raw payload is hand-built — +// exact digits ingest forwarded) excludes it. The row bytes go to ClickHouse's +// own parser untouched and the claim is compared through the strict integer +// cast, so no Go float is on the path at all. The raw payload is hand-built — // marshaling a Go float64 would already have destroyed the value this test is about. func TestHub_RowFilter_BigIntegerExact(t *testing.T) { t.Parallel() - reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ - {Name: "clicks", Columns: []discovery.Column{ - {Name: "tenant_id", Type: "UInt64"}, - {Name: "page", Type: "String"}, - }}, - }) p := &policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": {"viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"tenant_id": {Eq: new("{{ jwt.tenant }}")}}}}}, }, } - hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) + hub := chtypesHub(t, staticPolicy(p), nil, + chtypesTable("clicks", col("page", "String"), col("tenant_id", "UInt64"))) topic := topicOf("clicks") // Claims come from real signed tokens through the production middleware, so a @@ -1025,20 +1110,14 @@ func TestHub_RowFilter_BigIntegerExact(t *testing.T) { "the wire frame carries the exact digits, not a float64 rounding") } -// TestHub_RowFilter_TimestampInstantMatch: since #402, ingest canonicalizes -// DateTime/DateTime64 payload values to RFC 3339 UTC before publish, while policy -// authors write the ClickHouse-friendly zone-less spelling the query path wants. -// The row filter compares the two as instants through the discovery grammar (one -// parser shared with canonicalization), so the spellings agree; an operand the -// grammar can't read withholds the row. +// TestHub_RowFilter_TimestampInstantMatch: the wire carries ClickHouse's own +// rendering of a DateTime, and policy authors write the same zone-less spelling +// the query path wants. The filter compares them as instants because the row is +// parsed into the column's real storage before the predicate runs, so a +// different spelling of the same instant still matches; an operand the parser +// can't read withholds the row rather than guessing at it. func TestHub_RowFilter_TimestampInstantMatch(t *testing.T) { t.Parallel() - reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ - {Name: "clicks", Columns: []discovery.Column{ - {Name: "created_at", Type: "DateTime"}, - {Name: "page", Type: "String"}, - }}, - }) p := &policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -1048,21 +1127,27 @@ func TestHub_RowFilter_TimestampInstantMatch(t *testing.T) { }, }, } - hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) + hub := chtypesHub(t, staticPolicy(p), nil, + chtypesTable("clicks", col("created_at", "DateTime"), col("page", "String"))) topic := topicOf("clicks") sub := NewSubscriber(nil, nil) hub.Add(topic, "viewer", sub) - // The canonical wire spelling ingest publishes: same instant, different bytes. - hub.Broadcast(topic, rawEvent(t, "clicks", "t1", map[string]any{"created_at": "2026-06-21T04:00:00Z", "page": "/a"})) - f, _, _ := recvEvent(t, sub) - assert.NotEmpty(t, f.Data, "canonical payload matches the zone-less constant as an instant") + hub.Broadcast(topic, rawEvent(t, "clicks", "t1", map[string]any{"created_at": "2026-06-21 04:00:00", "page": "/a"})) + f, cols, _ := recvEvent(t, sub) + assert.NotEmpty(t, f.Data, "the wire rendering matches the zone-less constant") + + // A different spelling of the same instant matches too: the comparison is + // between parsed instants, not between bytes. + hub.Broadcast(topic, rawEvent(t, "clicks", "t2", map[string]any{"created_at": "2026-06-21T04:00:00Z", "page": "/a"})) + f, _ = recvEventCols(t, sub, cols) + assert.NotEmpty(t, f.Data) - hub.Broadcast(topic, rawEvent(t, "clicks", "t2", map[string]any{"created_at": "2026-06-21T04:00:01Z", "page": "/a"})) + hub.Broadcast(topic, rawEvent(t, "clicks", "t3", map[string]any{"created_at": "2026-06-21 04:00:01", "page": "/a"})) assertNoFrame(t, sub) - hub.Broadcast(topic, rawEvent(t, "clicks", "t3", map[string]any{"created_at": "not a timestamp", "page": "/a"})) + hub.Broadcast(topic, rawEvent(t, "clicks", "t4", map[string]any{"created_at": "not a timestamp", "page": "/a"})) assertNoFrame(t, sub) } @@ -1082,34 +1167,50 @@ func TestHub_RowFilterWithheldIncrementsMetric(t *testing.T) { otel.SetMeterProvider(savedMP) }) - hub := NewHub(staticPolicy(rowFilterPolicy()), nil, NewMetrics()) + hub := chtypesHub(t, staticPolicy(rowFilterPolicy()), NewMetrics(), clicksTable()) topic := topicOf("clicks") acme := NewSubscriber(map[string]any{"tenant": "acme"}, nil) globex := NewSubscriber(map[string]any{"tenant": "globex"}, nil) hub.Add(topic, "viewer", acme) hub.Add(topic, "viewer", globex) - raw := rawEvent(t, "clicks", "t", map[string]any{"tenant_id": "acme", "page": "/a"}) + raw := rawEvent(t, "clicks", "t", map[string]any{"tenant_id": "acme", "page": "/a", "secret": "x"}) hub.Broadcast(topic, raw) // delivered to acme, withheld from globex → 1 frames := hub.ReplayProjector(tenant.Default, "viewer", NewSubscriber(map[string]any{"tenant": "globex"}, nil))(raw) require.Empty(t, frames) // replay withhold → 2 + // A column the table does not have: a FAULT, not a filter verdict, and the + // label is the only thing that says so. It withholds from BOTH subscribers + // — nothing about this row is readable — → 4. + hub.Broadcast(topic, rawEvent(t, "clicks", "t", map[string]any{"page": "/a", "tenant_id": "acme", "dropped": "x"})) + + // An envelope without the filtered column (the inserting role does not + // write it): no verdict for either subscriber → 6. + hub.Broadcast(topic, rawEvent(t, "clicks", "t", map[string]any{"page": "/a"})) + var rm metricdata.ResourceMetrics require.NoError(t, reader.Collect(context.Background(), &rm)) - assert.Equal(t, int64(2), sumByName(rm, "wavehouse_sse_rows_withheld_total")) + assert.Equal(t, int64(6), sumByName(rm, "wavehouse_sse_rows_withheld_total")) + assert.Equal(t, int64(2), sumByNameAttr(rm, "wavehouse_sse_rows_withheld_total", "reason", ReasonFilter), + "the two claim mismatches are ordinary filtering") + assert.Equal(t, int64(2), sumByNameAttr(rm, "wavehouse_sse_rows_withheld_total", "reason", ReasonDrift), + "drift must be distinguishable from a policy decision") + assert.Equal(t, int64(2), sumByNameAttr(rm, "wavehouse_sse_rows_withheld_total", "reason", ReasonDecline), + "a filter over a column the event does not carry is no verdict at all") f, _, _ := recvEvent(t, acme) assert.NotEmpty(t, f.Data, "the entitled subscriber still gets the event") assertNoFrame(t, globex) } // BenchmarkBroadcast_RowFilteredFanout measures the per-subscriber cost a -// row-filtered role pays on the delivery hot path (#294/#353 vs #319): each -// subscriber's claims run through policy.Evaluate + RowVisible per event, where an -// unfiltered role shares one projection bucket-wide. Half the subscribers share the -// event's tenant (row visible), half don't (row withheld); either way each pays the -// per-subscriber evaluation, which is the cost under measurement. See #435 for the -// memoization follow-up this benchmark exists to arbitrate. +// row-filtered role pays on the delivery hot path (#294/#353 vs #319): the row is +// parsed once per event, then each subscriber's claims run through +// policy.Evaluate and one compiled-predicate evaluation, where an unfiltered role +// shares one projection bucket-wide. Half the subscribers share the event's +// tenant (row visible), half don't (row withheld); either way each pays the +// per-subscriber evaluation, which is the cost under measurement. See #435 for +// the memoization follow-up this benchmark exists to arbitrate. func BenchmarkBroadcast_RowFilteredFanout(b *testing.B) { topic := topicOf("clicks") raw := rawEvent(b, "clicks", "2026-06-26T00:00:00Z", @@ -1117,7 +1218,7 @@ func BenchmarkBroadcast_RowFilteredFanout(b *testing.B) { for _, n := range []int{100, 1_000, 10_000} { b.Run(fmt.Sprintf("subscribers=%d", n), func(b *testing.B) { - hub := NewHub(staticPolicy(rowFilterPolicy()), nil, nil) + hub := chtypesHub(b, staticPolicy(rowFilterPolicy()), nil, clicksTable()) subs := make([]*Subscriber, n) for i := range n { tenant := "acme" @@ -1221,6 +1322,30 @@ func sumByNameKind(rm metricdata.ResourceMetrics, name, kind string) int64 { return 0 } +// sumByNameAttr sums one counter's data points restricted to a single attribute +// value — the withheld counter's "reason" is a closed set, and the point of the +// label is that a fault does not read as a filter verdict. +func sumByNameAttr(rm metricdata.ResourceMetrics, name, key, want string) int64 { + var total int64 + for _, sm := range rm.ScopeMetrics { + for _, m := range sm.Metrics { + if m.Name != name { + continue + } + sum, ok := m.Data.(metricdata.Sum[int64]) + if !ok { + continue + } + for _, dp := range sum.DataPoints { + if v, found := dp.Attributes.Value(attribute.Key(key)); found && v.AsString() == want { + total += dp.Value + } + } + } + } + return total +} + // sumByName totals all datapoints of an Int64 sum instrument across kinds. func sumByName(rm metricdata.ResourceMetrics, name string) int64 { for _, sm := range rm.ScopeMetrics { @@ -1495,7 +1620,7 @@ func fixedRegistry(reg *discovery.SchemaRegistry) RegistrySource { // TestHub_RegistrySourceYieldingNilIsNoSchema: a tenant with no registry — // not served, or its registry not built yet — reads exactly like a hub with -// no registry at all: nothing to announce, every column opaque. +// no registry at all: nothing to announce. func TestHub_RegistrySourceYieldingNilIsNoSchema(t *testing.T) { t.Parallel() var asked []tenant.ID @@ -1505,7 +1630,8 @@ func TestHub_RegistrySourceYieldingNilIsNoSchema(t *testing.T) { }, nil) _, ok := hub.SubscribeSchemaFrame("acme", "clicks", "viewer", NewSubscriber(nil, nil)) assert.False(t, ok) - assert.Nil(t, hub.columnSpecs("globex", "clicks")) + _, ok = hub.SubscribeSchemaFrame("globex", "clicks", "viewer", NewSubscriber(nil, nil)) + assert.False(t, ok) assert.Equal(t, []tenant.ID{"acme", "globex"}, asked, "the source is asked for the tenant the lookup names") } diff --git a/internal/stream/metrics.go b/internal/stream/metrics.go index c49c4bd1..9c4ed3b6 100644 --- a/internal/stream/metrics.go +++ b/internal/stream/metrics.go @@ -45,7 +45,7 @@ func NewMetrics() *Metrics { dropped, _ := meter.Int64Counter("wavehouse_sse_dropped_frames_total", metric.WithDescription("SSE frames dropped to a full subscriber queue (slow consumer)")) withheld, _ := meter.Int64Counter("wavehouse_sse_rows_withheld_total", - metric.WithDescription("Event rows withheld from a subscriber by the role's row-level-security filter (including fail-closed evaluations)")) + metric.WithDescription("Event rows withheld from a subscriber by the role's row-level-security filter, labelled by why (including fail-closed evaluations)")) return &Metrics{active: active, duration: duration, frames: frames, bytes: bytes, dropped: dropped, withheld: withheld} } @@ -89,11 +89,22 @@ func (m *Metrics) FrameDropped(kind string) { // RowWithheld records one event row withheld from one subscriber (live or replay) // by the role's row-level-security filter, including fail-closed evaluations. // Labeled by table and role (policy-bounded, not data-bounded) so an operator can -// tell "no matching rows" from "a misconfigured filter withholding everything". -func (m *Metrics) RowWithheld(table, role string) { +// tell "no matching rows" from "a misconfigured filter withholding everything", +// and by reason (a Reason* constant, a closed set) so the cases that are a FAULT +// rather than a filter verdict are visible on their own: `unavailable` means the +// type layer has no compiled schema for the tenant's table, so every +// row-filtered subscriber is dark until it does; `drift` means events arrive +// under a column list the table's current generation cannot read; `error` means +// ClickHouse raised evaluating the predicate over the row; and `decline` means +// no verdict was reached — a filter that does not compile, a row that does not +// parse, or a filter on a column the inserting role did not write. Only +// `filter` is a policy decision. +func (m *Metrics) RowWithheld(table, role, reason string) { if m == nil { return } - m.withheld.Add(context.Background(), 1, - metric.WithAttributes(attribute.String("table", table), attribute.String("role", role))) + m.withheld.Add(context.Background(), 1, metric.WithAttributes( + attribute.String("table", table), + attribute.String("role", role), + attribute.String("reason", reason))) } diff --git a/internal/stream/roweval.go b/internal/stream/roweval.go new file mode 100644 index 00000000..d8784a43 --- /dev/null +++ b/internal/stream/roweval.go @@ -0,0 +1,211 @@ +package stream + +import ( + "encoding/json" + "errors" + "log/slog" + "sync" + + "github.com/Wave-RF/WaveHouse/internal/policy" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" +) + +// Reasons a row was withheld, the "reason" label on +// wavehouse_sse_rows_withheld_total. The first three are the type layer's own +// verdict classes, aliased so the two packages cannot drift on the spelling; the +// last two are this package's, for the cases where no verdict was ever reached. +const ( + // ReasonFilter: the role's predicate answered false — the ordinary case, and + // the only one that is not a symptom of something being wrong. It is also + // the answer for a predicate whose claim was unresolvable, which matches no + // row without the type layer compiling anything (the query path's `1 = 0`). + ReasonFilter = typelayer.ReasonFilter + // ReasonError: ClickHouse evaluated the predicate over this row and raised. + // Either side of the comparison can cause it, and both are the policy + // author's to fix: a stored value the expression cannot read, or a filter + // constant the column's type cannot read (a claim rendering as "abc" or + // "1.5" against a numeric column answers code 53 per row). + ReasonError = typelayer.ReasonError + // ReasonDecline: no verdict was reached. The expression would not compile + // for this generation, chtypes would not answer for the row (one that does + // not parse lands here), or the predicate reads a column the event does not + // carry (see engineRowView.Visible). Withheld, like every answer that is not + // a definite true. + ReasonDecline = typelayer.ReasonDecline + // ReasonUnavailable: no compiled schema can answer for this tenant's table + // — the tenant is not bound yet, its server line has no artifact, a compile + // refusal, or no engine wired at all. Nothing about the row; every + // row-filtered subscriber of the table is affected until it is fixed. + ReasonUnavailable = "unavailable" + // ReasonDrift: the event's column list names a column the table's current + // generation does not export (dropped or renamed since), or names one twice, + // so its positional row cannot be read at all. Expected briefly after a + // schema change, alarming if it persists. + ReasonDrift = "drift" +) + +// RowEvaluator is the one place a row's visibility under a role's row-filter is +// decided. Prepare is called ONCE per event — parsing the positional row is the +// expensive half — and the returned view answers for each subscriber's resolved +// permissions. The interface exists because that ratio (one parse, K visibility +// questions) is the whole shape of the hot path, and because it lets the tests +// drive admission from a stub instead of a compiled schema; the production +// implementation is engineEvaluator below. +// +// A nil RowEvaluator on the Hub means the fail-closed default (see +// Hub.rowEvaluator), never "everything visible". +type RowEvaluator interface { + // Prepare reads one event's row for tenant id's table. columns is the + // envelope's column list — the inserting role's, which may be narrower than + // the table — and row the positional JSON array as published. An error + // means no subscriber of a row-filtered role may see this row; the Hub + // labels the withhold with WithheldReason(err). + Prepare(id tenant.ID, table string, columns []string, row json.RawMessage) (RowView, error) +} + +// RowView is one prepared event row, reusable across every subscriber of every +// row-filtered role on that event. Close must be called once the fan-out is +// done: the parsed row holds native memory and a schema handle. +type RowView interface { + // Visible reports whether perms admit this row, and when they do not, the + // Reason* label the withheld metric wants. reason is "" when visible. + Visible(perms *policy.ResolvedPermissions) (visible bool, reason string) + Close() +} + +// NewRowEvaluator builds the production evaluator over the process's type +// layer: the row is parsed by ClickHouse's own parser, under the column list it +// was published with, and the role's predicate is evaluated by ClickHouse's own +// expression engine, so the stream's verdict is the one the query path's WHERE +// reaches over the stored row. +// +// A nil engine is a valid argument and withholds every row-filtered row — the +// same answer the Hub's unwired default gives — so a boot path that has no +// Engine fails closed rather than silently unfiltered. +func NewRowEvaluator(eng *typelayer.Engine) RowEvaluator { + return &engineEvaluator{types: eng} +} + +// engineEvaluator is the RowEvaluator every production path uses: a +// typelayer.Engine resolving the tenant's table, parsing the row and evaluating +// the role's predicates on it. +type engineEvaluator struct { + types *typelayer.Engine + // unwired reports the missing engine once rather than once per event. The + // condition is a boot-time wiring mistake, so the first line says everything + // the next million would. + unwired sync.Once +} + +// errNoEngine is the cause behind an unwired evaluator's withholds. It is not a +// verdict about the row: no row of any row-filtered role can be evaluated. +var errNoEngine = errors.New("no type engine is wired into the stream hub") + +func (e *engineEvaluator) Prepare(id tenant.ID, table string, columns []string, row json.RawMessage) (RowView, error) { + if e.types == nil { + e.unwired.Do(func() { + slog.Error("row-level security cannot be evaluated: no type engine is wired into the stream hub; "+ + "every row is withheld from every subscriber of a row-filtered role until one is", + "tenant", id, "table", table) + }) + return nil, &withheldError{reason: ReasonUnavailable, err: errNoEngine} + } + tbl, err := e.types.Table(id, table) + if err != nil { + return nil, classifyPrepare(err) + } + parsed, err := tbl.ParseRow(columns, row) + if err != nil { + tbl.Release() + return nil, classifyPrepare(err) + } + carried := make(map[string]struct{}, len(columns)) + for _, c := range columns { + carried[c] = struct{}{} + } + return &engineRowView{tbl: tbl, row: parsed, carried: carried}, nil +} + +// engineRowView holds the table handle for as long as the parsed row lives: the +// row is owned by the compiled schema, so releasing the handle first would leave +// a rebind free of memory the fan-out is still reading. +type engineRowView struct { + tbl *typelayer.Table + row *typelayer.Row + // carried is the envelope's column list: the values the inserting role + // actually supplied. + carried map[string]struct{} +} + +// Visible evaluates perms' row filter over the parsed row. +// +// A predicate over a column the event does not carry withholds the row from +// this subscriber, labelled ReasonDecline: the stream declines to judge a value +// the row never supplied. The parse does hold a value there — the DEFAULT, or +// the MATERIALIZED expression, computed when the stream parsed the row — but +// the stored row's value is computed again when the worker inserts, and a +// DEFAULT such as now(), rand() or generateUUIDv4() lands differently, so a +// verdict on the stream's copy could admit a row the query path's WHERE +// excludes. Withholding costs availability for that one role/column pairing, +// never confidentiality. Decline rather than filter because it is not a policy +// verdict: the reading role filters on a column the inserting role does not +// write, which an operator should be able to see. +func (v *engineRowView) Visible(perms *policy.ResolvedPermissions) (bool, string) { + preds, ok := perms.Predicates() + if !ok { + // A denied grant, or one resolved for INSERT: no row is admissible. Same + // answer the query path gives by never running the SELECT at all. + return false, ReasonFilter + } + if len(preds) == 0 { + return true, "" + } + for _, p := range preds { + if _, carried := v.carried[p.Column]; !carried { + return false, ReasonDecline + } + } + return v.row.VisibleWithReason(preds) +} + +func (v *engineRowView) Close() { + v.row.Close() + v.tbl.Release() +} + +// withheldError carries the metric label alongside the cause, so the Hub does +// not have to re-derive from an error string what the evaluator already knew. +type withheldError struct { + reason string + err error +} + +func (e *withheldError) Error() string { return e.err.Error() } +func (e *withheldError) Unwrap() error { return e.err } + +// classifyPrepare names WHY no view could be prepared. The causes are +// operationally different — a missing artifact or an unbound tenant is an +// estate problem, drift is a schema change in flight — and they are +// indistinguishable in the metric without the label. +func classifyPrepare(err error) error { + switch { + case typelayer.IsUnavailable(err): + return &withheldError{reason: ReasonUnavailable, err: err} + case errors.Is(err, typelayer.ErrColumnsDrift): + return &withheldError{reason: ReasonDrift, err: err} + default: + return &withheldError{reason: ReasonError, err: err} + } +} + +// WithheldReason maps a Prepare error to its metric label. An evaluator that +// returns a plain error (a test double, a future implementation) reads as +// "error" rather than losing the withhold. +func WithheldReason(err error) string { + var w *withheldError + if errors.As(err, &w) { + return w.reason + } + return ReasonError +} diff --git a/internal/stream/roweval_test.go b/internal/stream/roweval_test.go index 3adea1bb..c53c919a 100644 --- a/internal/stream/roweval_test.go +++ b/internal/stream/roweval_test.go @@ -1,27 +1,53 @@ package stream import ( + "encoding/json" + "errors" + "fmt" "testing" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) -// recordingEvaluator answers every row the same way and counts the calls, so a -// test can prove both delivery paths reach row-level security through the seam. +// recordingEvaluator answers every row the same way and counts what the Hub +// does, so a test can prove both delivery paths reach row-level security through +// the seam — and that each path parses once and closes what it parsed. type recordingEvaluator struct { - visible bool - calls int + visible bool + prepErr error + prepares int + calls int + closes int + tenants []tenant.ID } -func (e *recordingEvaluator) Visible(*policy.ResolvedPermissions, map[string]any, map[string]policy.ColumnSpec) bool { - e.calls++ - return e.visible +func (e *recordingEvaluator) Prepare(id tenant.ID, _ string, _ []string, _ json.RawMessage) (RowView, error) { + e.prepares++ + e.tenants = append(e.tenants, id) + if e.prepErr != nil { + return nil, e.prepErr + } + return &recordingView{e: e}, nil +} + +type recordingView struct{ e *recordingEvaluator } + +func (v *recordingView) Visible(*policy.ResolvedPermissions) (bool, string) { + v.e.calls++ + if v.e.visible { + return true, "" + } + return false, ReasonFilter } +func (v *recordingView) Close() { v.e.closes++ } + // filteredPolicy grants "viewer" a row-filter, which is what puts the Hub on // the per-subscriber admission path in the first place. func filteredPolicy() *policy.Policy { @@ -45,8 +71,8 @@ func TestHub_RowEvaluatorSeam_LiveBroadcast(t *testing.T) { t.Parallel() // The tenant is chosen per case so each subtest builds the scenario its name // describes. With both cases sending a row the predicate withholds, the - // withhold case proved nothing — `seam.Visible(…) || perms.RowVisible(…)` - // would have passed it, since the policy withheld the row anyway. + // withhold case proved nothing — a seam consulted only as a veto would have + // passed it, since the policy withheld the row anyway. for _, tt := range []struct { name string visible bool @@ -71,6 +97,8 @@ func TestHub_RowEvaluatorSeam_LiveBroadcast(t *testing.T) { map[string]any{"tenant_id": tt.tenantID, "page": "/a"})) assert.Equal(t, 1, eval.calls, "the live path must consult the seam") + assert.Equal(t, 1, eval.prepares, "the row is parsed once for the event") + assert.Equal(t, 1, eval.closes, "and released after the fan-out") if tt.visible { f, _, _ := recvEvent(t, sub) assert.NotEmpty(t, f.Data) @@ -81,6 +109,84 @@ func TestHub_RowEvaluatorSeam_LiveBroadcast(t *testing.T) { } } +// TestHub_RowEvaluatorSeam_PreparesOncePerEvent: parsing the row is the +// expensive half of a row-filter decision, so it happens once per EVENT, not +// once per subscriber or once per role — that ratio is the whole reason the seam +// is split into Prepare and Visible. +func TestHub_RowEvaluatorSeam_PreparesOncePerEvent(t *testing.T) { + t.Parallel() + p := filteredPolicy() + // A second filtered role on the same table: the parse must be shared across + // roles too, not just across a role's subscribers. + tmpl := "{{ jwt.tenant }}" + p.Tables["clicks"]["editor"] = policy.RolePermissions{Select: &policy.SelectPermissions{ + Filter: map[string]policy.Filter{"tenant_id": {Eq: &tmpl}}, + }} + + eval := &recordingEvaluator{visible: true} + hub := NewHub(staticPolicy(p), nil, nil) + hub.RowEvaluator = eval + for _, role := range []string{"viewer", "viewer", "viewer", "editor"} { + hub.Add(topicOf("clicks"), role, NewSubscriber(map[string]any{"tenant": "t1"}, nil)) + } + + hub.Broadcast(topicOf("clicks"), rawEvent(t, "clicks", "2026-06-26T00:00:00Z", + map[string]any{"tenant_id": "t1", "page": "/a"})) + + assert.Equal(t, 1, eval.prepares, "one parse serves every role and subscriber") + assert.Equal(t, 4, eval.calls, "visibility is still decided per subscriber") + assert.Equal(t, 1, eval.closes) +} + +// TestHub_RowEvaluatorSeam_PreparesForTheTopicsTenant: the row is prepared +// against the tenant whose topic carried it on the live path, and against the +// connection's tenant on a gap-fill — the type layer keeps one schema per +// tenant, and two tenants' tables of one name are different tables. +func TestHub_RowEvaluatorSeam_PreparesForTheTopicsTenant(t *testing.T) { + t.Parallel() + eval := &recordingEvaluator{visible: true} + hub := NewHub(staticPolicy(filteredPolicy()), nil, nil) + hub.RowEvaluator = eval + acme := topicOf("clicks") + acme.Tenant = "acme" + hub.Add(acme, "viewer", NewSubscriber(map[string]any{"tenant": "t1"}, nil)) + + raw := rawEvent(t, "clicks", "2026-06-26T00:00:00Z", map[string]any{"tenant_id": "t1", "page": "/a"}) + hub.Broadcast(acme, raw) + hub.ReplayProjector("globex", "viewer", NewSubscriber(map[string]any{"tenant": "t1"}, nil))(raw) + + assert.Equal(t, []tenant.ID{"acme", "globex"}, eval.tenants) +} + +// TestHub_RowEvaluatorSeam_PrepareError_WithholdsFilteredRolesOnly: when the row +// cannot be prepared at all — no compiled schema, a column list the table +// cannot have — every row-filtered subscriber is withheld. Roles WITHOUT a +// row-filter never asked the type layer anything, so they must be unaffected: a +// table whose schema handle is briefly unavailable must not black out the +// streams that do not depend on it. +func TestHub_RowEvaluatorSeam_PrepareError_WithholdsFilteredRolesOnly(t *testing.T) { + t.Parallel() + p := filteredPolicy() + p.Tables["clicks"]["public"] = policy.RolePermissions{Select: &policy.SelectPermissions{}} + + eval := &recordingEvaluator{visible: true, prepErr: errors.New("boom")} + hub := NewHub(staticPolicy(p), nil, nil) + hub.RowEvaluator = eval + + filtered := NewSubscriber(map[string]any{"tenant": "t1"}, nil) + unfiltered := NewSubscriber(nil, nil) + hub.Add(topicOf("clicks"), "viewer", filtered) + hub.Add(topicOf("clicks"), "public", unfiltered) + + hub.Broadcast(topicOf("clicks"), rawEvent(t, "clicks", "2026-06-26T00:00:00Z", + map[string]any{"tenant_id": "t1", "page": "/a"})) + + assertNoFrame(t, filtered) + f, _, _ := recvEvent(t, unfiltered) + assert.NotEmpty(t, f.Data, "an unfiltered role never consults the type layer, so it is unaffected") + assert.Zero(t, eval.calls, "a row that could not be prepared is never asked about") +} + // TestHub_RowEvaluatorSeam_Replay: the gap-fill path goes through the same seam // as the live path, so the two can't drift on how row visibility is decided. func TestHub_RowEvaluatorSeam_Replay(t *testing.T) { @@ -96,24 +202,228 @@ func TestHub_RowEvaluatorSeam_Replay(t *testing.T) { assert.Empty(t, frames, "the seam's verdict decides replay too") assert.Equal(t, 1, eval.calls) + assert.Equal(t, 1, eval.prepares) + assert.Equal(t, 1, eval.closes, "a gap-fill of thousands of events must not accumulate parsed rows") } -// TestHub_DefaultRowEvaluator_WhenUnwired: an un-wired Hub still enforces -// row-level security. A nil seam must never read as "everything is visible". +// TestHub_DefaultRowEvaluator_WhenUnwired: an un-wired Hub must never read as +// "everything is visible". With the type layer behind the seam there is nothing +// left in-process that can evaluate a filter, so the only safe default is to +// withhold every row of every row-filtered role — including the rows the +// predicate would have admitted, which is what makes this a real assertion +// rather than a restatement of the policy. func TestHub_DefaultRowEvaluator_WhenUnwired(t *testing.T) { t.Parallel() hub := NewHub(staticPolicy(filteredPolicy()), nil, nil) require.Nil(t, hub.RowEvaluator) - assert.IsType(t, policyRowEvaluator{}, hub.rowEvaluator()) + assert.IsType(t, &engineEvaluator{}, hub.rowEvaluator(), + "the default must be the engine-backed evaluator with no engine, not a permissive stub") sub := NewSubscriber(map[string]any{"tenant": "t1"}, nil) hub.Add(topicOf("clicks"), "viewer", sub) + hub.Broadcast(topicOf("clicks"), rawEvent(t, "clicks", "2026-06-26T00:00:00Z", map[string]any{"tenant_id": "t2", "page": "/a"})) assertNoFrame(t, sub) + // The row the filter WOULD admit is withheld too: no engine, no verdict. hub.Broadcast(topicOf("clicks"), rawEvent(t, "clicks", "2026-06-26T00:00:01Z", map[string]any{"tenant_id": "t1", "page": "/a"})) - f, _, _ := recvEvent(t, sub) - assert.NotEmpty(t, f.Data) + assertNoFrame(t, sub) + + assert.Empty(t, hub.ReplayProjector(tenant.Default, "viewer", sub)(rawEvent(t, "clicks", "2026-06-26T00:00:02Z", + map[string]any{"tenant_id": "t1", "page": "/a"})), "replay fails closed on the same grounds") +} + +// TestNewRowEvaluator_NilEngineFailsClosed: the constructor accepts a nil Engine +// (a process with no type layer still has to build a Hub) and answers the same +// way the unwired default does, with the reason an operator needs. The reason +// matters: `unavailable` says the estate is broken, where `filter` would say +// the policy is working. +func TestNewRowEvaluator_NilEngineFailsClosed(t *testing.T) { + t.Parallel() + view, err := NewRowEvaluator(nil).Prepare(tenant.Default, "clicks", []string{"page"}, json.RawMessage(`["/a"]`)) + require.Error(t, err) + assert.Nil(t, view) + assert.Equal(t, ReasonUnavailable, WithheldReason(err)) +} + +// TestWithheldReason_ClassifiesTypeLayerErrors: the fault causes are +// operationally different and must not collapse into one label. An error from +// another evaluator that names no reason reads as "error" rather than being +// lost. +func TestWithheldReason_ClassifiesTypeLayerErrors(t *testing.T) { + t.Parallel() + assert.Equal(t, ReasonUnavailable, + WithheldReason(classifyPrepare(&typelayer.Unavailable{Tenant: tenant.Default, Table: "clicks", Cause: "no artifact"}))) + assert.Equal(t, ReasonDrift, + WithheldReason(classifyPrepare(fmt.Errorf("wrapped: %w", typelayer.ErrColumnsDrift)))) + assert.Equal(t, ReasonError, WithheldReason(classifyPrepare(errors.New("unparseable row")))) + assert.Equal(t, ReasonError, WithheldReason(errors.New("an evaluator that names no reason"))) +} + +// TestEngineEvaluator_TenantsAreIndependent: the production evaluator resolves +// the EVENT's tenant's table. A tenant the type layer has not bound is +// unavailable on its own, while a bound tenant's row of the same table name is +// judged normally. +func TestEngineEvaluator_TenantsAreIndependent(t *testing.T) { + t.Parallel() + eng := typelayer.TestEngine(t, clicksTable()) + eng.Bind("acme", typelayer.TestServerVersion, "UTC", []*discovery.TableSchema{clicksTable()}) + eval := NewRowEvaluator(eng) + cols := []string{"page", "secret", "tenant_id"} + row := json.RawMessage(`["/a","x","acme"]`) + perms := policy.Evaluate(filteredPolicy(), "viewer", "clicks", "select", map[string]any{"tenant": "acme"}) + + for _, id := range []tenant.ID{tenant.Default, "acme"} { + view, err := eval.Prepare(id, "clicks", cols, row) + require.NoError(t, err, id) + visible, reason := view.Visible(perms) + view.Close() + assert.True(t, visible, "%s: %s", id, reason) + } + + _, err := eval.Prepare("globex", "clicks", cols, row) + require.Error(t, err) + assert.Equal(t, ReasonUnavailable, WithheldReason(err), "an unbound tenant is unavailable, not a verdict") + + _, err = eval.Prepare("acme", "views", cols, row) + assert.Equal(t, ReasonUnavailable, WithheldReason(err), "a table the tenant does not have is unavailable") +} + +// roleWidthTable is a table some roles may write only part of. "region" has a +// DEFAULT the stream's parse fills in, which is exactly why a filter over it +// must still withhold when the event does not carry it. +func roleWidthTable() *discovery.TableSchema { + return chtypesTable("clicks", + col("page", "String"), + col("secret", "String"), + discovery.Column{Name: "region", Type: "String", DefaultKind: "DEFAULT", DefaultExpression: "'eu'"}, + col("tenant_id", "String")) +} + +// TestHub_RowFilter_RoleWidthEnvelope: ingest publishes each row with the +// INSERTING role's columns, so a role that may not write "secret" or "region" +// publishes a narrower envelope than the table. Such a row is evaluated, not +// withheld as drift: it reaches the subscriber whose claim matches and is +// withheld from the one whose claim does not. +func TestHub_RowFilter_RoleWidthEnvelope(t *testing.T) { + t.Parallel() + hub := chtypesHub(t, staticPolicy(rowFilterPolicy()), nil, roleWidthTable()) + acme := NewSubscriber(map[string]any{"tenant": "acme"}, nil) + globex := NewSubscriber(map[string]any{"tenant": "globex"}, nil) + hub.Add(topicOf("clicks"), "viewer", acme) + hub.Add(topicOf("clicks"), "viewer", globex) + + hub.Broadcast(topicOf("clicks"), rawEventCols(t, "clicks", "t1", []string{"page", "tenant_id"}, + map[string]any{"page": "/a", "tenant_id": "acme"})) + + _, cols, row := recvEvent(t, acme) + assert.Equal(t, []string{"page"}, cols) + assert.Equal(t, "/a", row["page"]) + assertNoFrame(t, globex) + + frames := hub.ReplayProjector(tenant.Default, "viewer", NewSubscriber(map[string]any{"tenant": "acme"}, nil))( + rawEventCols(t, "clicks", "t1", []string{"page", "tenant_id"}, map[string]any{"page": "/a", "tenant_id": "acme"})) + assert.Len(t, frames, 2, "replay evaluates the narrower envelope the same way") +} + +// TestHub_RowFilter_PermutedEnvelope: the row is positional against the +// envelope's own list, whatever its order — the k-th cell is the k-th named +// column, full width or narrower. The announced list follows the envelope, so +// the client zips each value under its own name. +func TestHub_RowFilter_PermutedEnvelope(t *testing.T) { + t.Parallel() + p := rowFilterPolicy() + p.Tables["clicks"]["auditor"] = policy.RolePermissions{Select: &policy.SelectPermissions{ + Filter: map[string]policy.Filter{"tenant_id": {Eq: new("{{ jwt.tenant }}")}}, + }} + hub := chtypesHub(t, staticPolicy(p), nil, roleWidthTable()) + + for _, cols := range [][]string{ + {"tenant_id", "region", "secret", "page"}, + {"tenant_id", "page"}, + } { + acme := NewSubscriber(map[string]any{"tenant": "acme"}, nil) + globex := NewSubscriber(map[string]any{"tenant": "globex"}, nil) + hub.Add(topicOf("clicks"), "auditor", acme) + hub.Add(topicOf("clicks"), "auditor", globex) + + hub.Broadcast(topicOf("clicks"), rawEventCols(t, "clicks", "t1", cols, + map[string]any{"page": "/a", "tenant_id": "acme", "secret": "s", "region": "us"})) + + _, announced, row := recvEvent(t, acme) + assert.Equal(t, cols, announced, "the announcement follows the envelope's order") + assert.Equal(t, "/a", row["page"]) + assert.Equal(t, "acme", row["tenant_id"]) + assertNoFrame(t, globex) + + hub.Remove(topicOf("clicks"), "auditor", acme) + hub.Remove(topicOf("clicks"), "auditor", globex) + } +} + +// TestHub_RowFilter_FilterOnAbsentColumnWithholds: a filter over a column the +// event does not carry fails closed for that subscriber, even where the +// table's DEFAULT ("eu") would satisfy it — the stored row's DEFAULT is +// computed again at insert, and a stream verdict on its own copy could admit +// a row the query path excludes. The reason is decline, not filter: it is no +// policy verdict. The control delivers the same row once it carries the value. +func TestHub_RowFilter_FilterOnAbsentColumnWithholds(t *testing.T) { + t.Parallel() + p := &policy.Policy{Tables: map[string]policy.TablePolicy{ + "clicks": {"viewer": {Select: &policy.SelectPermissions{ + Filter: map[string]policy.Filter{"region": {Eq: new("eu")}}, + }}}, + }} + eng := typelayer.TestEngine(t, roleWidthTable()) + hub := NewHub(staticPolicy(p), nil, nil) + hub.RowEvaluator = NewRowEvaluator(eng) + sub := NewSubscriber(nil, nil) + hub.Add(topicOf("clicks"), "viewer", sub) + + narrow := []string{"page", "tenant_id"} + hub.Broadcast(topicOf("clicks"), rawEventCols(t, "clicks", "t1", narrow, + map[string]any{"page": "/a", "tenant_id": "acme"})) + assertNoFrame(t, sub) + + view, err := NewRowEvaluator(eng).Prepare(tenant.Default, "clicks", narrow, json.RawMessage(`["/a","acme"]`)) + require.NoError(t, err) + visible, reason := view.Visible(policy.Evaluate(p, "viewer", "clicks", "select", nil)) + view.Close() + assert.False(t, visible) + assert.Equal(t, ReasonDecline, reason) + + hub.Broadcast(topicOf("clicks"), rawEventCols(t, "clicks", "t2", []string{"page", "region"}, + map[string]any{"page": "/b", "region": "eu"})) + _, _, row := recvEvent(t, sub) + assert.Equal(t, "/b", row["page"]) +} + +// TestHub_RowFilter_UnknownColumnIsDrift: an envelope naming a column the +// table does not have (dropped or renamed since it was published) cannot be +// read at all, so it is withheld from every row-filtered subscriber as drift. +// A role without a row filter never asks the type layer and still gets it. +func TestHub_RowFilter_UnknownColumnIsDrift(t *testing.T) { + t.Parallel() + p := rowFilterPolicy() + p.Tables["clicks"]["public"] = policy.RolePermissions{Select: &policy.SelectPermissions{}} + eng := typelayer.TestEngine(t, roleWidthTable()) + hub := NewHub(staticPolicy(p), nil, nil) + hub.RowEvaluator = NewRowEvaluator(eng) + acme := NewSubscriber(map[string]any{"tenant": "acme"}, nil) + public := NewSubscriber(nil, nil) + hub.Add(topicOf("clicks"), "viewer", acme) + hub.Add(topicOf("clicks"), "public", public) + + cols := []string{"page", "tenant_id", "bogus"} + hub.Broadcast(topicOf("clicks"), rawEventCols(t, "clicks", "t1", cols, + map[string]any{"page": "/a", "tenant_id": "acme", "bogus": "x"})) + assertNoFrame(t, acme) + _, announced, _ := recvEvent(t, public) + assert.Equal(t, cols, announced) + + _, err := NewRowEvaluator(eng).Prepare(tenant.Default, "clicks", cols, json.RawMessage(`["/a","acme","x"]`)) + require.Error(t, err) + assert.Equal(t, ReasonDrift, WithheldReason(err)) } diff --git a/tests/e2e/sdk/streaming.test.ts b/tests/e2e/sdk/streaming.test.ts index 45d4e38b..d0a5f51c 100644 --- a/tests/e2e/sdk/streaming.test.ts +++ b/tests/e2e/sdk/streaming.test.ts @@ -20,7 +20,7 @@ describe("Streaming", () => { // Explicitly allow the 'anon' role to SELECT (stream) from this suite's tables. // 'scoped' additionally carries a per-subscriber row filter — streamed rows are // limited to the caller's own country claim — so the SSE fan-out exercises the - // row-level-security path (ResolvedPermissions.RowVisible) end to end, not just + // row-level-security path (the type layer's compiled filters) end to end, not just // column projection. Object.assign(publicPolicy.tables[T.clicks], { anon: { select: { allow_columns: ["*"] } }, @@ -31,8 +31,8 @@ describe("Streaming", () => { }, }, // 'metered' carries a numeric literal bound, so the SSE fan-out exercises - // the storage-domain numeric comparison (canonical decimal + integer range - // gate) end to end, not just String equality scoping. + // the storage-domain numeric comparison (the strict integer cast on an + // integer column) end to end, not just String equality scoping. metered: { select: { allow_columns: ["*"], @@ -105,9 +105,11 @@ describe("Streaming", () => { it("row DateTime columns arrive canonicalized, matching /v1/query (#372)", async () => { // Ingest spells the row timestamp with an offset; the wire form everywhere - // downstream must be canonical RFC 3339 UTC, so the SSE frame and the - // /v1/query rendering of the same stored instant are byte-identical — the - // query/stream clock drift #372 reported. + // downstream is ClickHouse's OWN rendering of the stored instant, so the + // SSE frame and the /v1/query rendering are byte-identical — the + // query/stream clock drift #372 reported. Since the row is produced by the + // server's writer at validation time, that identity now holds by + // construction rather than by a canonicalizer agreeing with the server. const whPublic = publicClient(); const whAuth = dataClient(); const receivedEvents: any[] = []; @@ -133,7 +135,9 @@ describe("Streaming", () => { await waitForCondition(() => receivedEvents.some((e) => e.data?.event_id === id), 10_000); const frame = receivedEvents.find((e) => e.data?.event_id === id); - expect(frame?.data.received_timestamp).toBe("2026-06-21T04:00:00.123Z"); + // The column's own rendering: "YYYY-MM-DD hh:mm:ss.SSS" in its zone + // (DateTime64(3) here), not RFC 3339 with a Z. + expect(frame?.data.received_timestamp).toBe("2026-06-21 04:00:00.123"); // The ClickHouse insert is async behind the stream event — poll the query // path until the row lands, then compare the two renderings. @@ -298,9 +302,9 @@ describe("Streaming", () => { it("evaluates a numeric row filter in the column's storage domain (UInt32 threshold)", async () => { // 'metered' scopes delivery to duration_ms > 100 over a UInt32 column: the - // constant routes through the canonical-decimal reading and the integer - // range gate, and both operands compare in the column's storage domain — - // the #381 storage-domain path, pinned end to end on SSE. + // constant is read through the strict integer cast for the column's type, + // and both operands compare in the column's storage domain — the #381 + // storage-domain path, pinned end to end on SSE. const inserter = dataClient(); const client = authClient("metered"); const events: any[] = []; From 0537e89b0b373bb1e3abf28472e6492c5011eddf Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:04:22 -0400 Subject: [PATCH 11/70] feat(ingest)!: judge every body with ClickHouse's own parser Ingest hands the request body to the tenant's chtypes type layer in one parse (Table.IngestWith through the role's projection) instead of decoding, validating, checking, injecting and re-encoding records in Go. The verdicts feed the existing windowed dedupe (reserve, publish, commit, release) and the envelope carries ClickHouse's exported row with the role's wire columns. - content_type.go (was record_reader.go): text/csv and text/tab-separated-values, with RFC 4180's header parameter three ways (present, absent, auto-detect); any other header value is a 415. - ingest_framing.go: depth-1 comma reframing of a compact JSON array so one bad record does not cost its siblings, and the dedupe id read by position from the exported row. A null id cell is missing, like an absent one. - Error bodies keep the string code class. ClickHouse's numeric code travels as exception_code: per record, and on a whole-request header refusal alongside code clickhouse.rejected. - A tenant or table the type layer cannot judge is a 503 with Retry-After: 5, decided before the body is read, with a generic body and the cause in the log. - Column policy is the role's compiled schema, so a denied column is ClickHouse's 117 (400) instead of a gateway 403; parse errors now precede check errors. - The RecordValidator/InsertChecker seams are gone. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/api/bufpool.go | 8 +- .../api/{record_reader.go => content_type.go} | 364 +++---- internal/api/errors.go | 13 +- internal/api/ingest.go | 910 +++++++++++------- internal/api/ingest_formats_test.go | 411 ++++++++ internal/api/ingest_framing.go | 191 ++++ internal/api/ingest_framing_test.go | 173 ++++ internal/api/ingest_retention_test.go | 4 +- internal/api/ingest_seams.go | 82 -- internal/api/ingest_seams_test.go | 246 ----- internal/api/ingest_test.go | 839 +++++++++++----- internal/api/ingest_unavailable_test.go | 202 ++++ internal/api/ingest_window_test.go | 4 +- 13 files changed, 2323 insertions(+), 1124 deletions(-) rename internal/api/{record_reader.go => content_type.go} (59%) create mode 100644 internal/api/ingest_formats_test.go create mode 100644 internal/api/ingest_framing.go create mode 100644 internal/api/ingest_framing_test.go delete mode 100644 internal/api/ingest_seams.go delete mode 100644 internal/api/ingest_seams_test.go create mode 100644 internal/api/ingest_unavailable_test.go diff --git a/internal/api/bufpool.go b/internal/api/bufpool.go index 7ff917da..f5b58cee 100644 --- a/internal/api/bufpool.go +++ b/internal/api/bufpool.go @@ -25,10 +25,10 @@ func getBodyBuffer() *bytes.Buffer { } // putBodyBuffer returns buf to the pool unless it outgrew -// maxPooledBufferBytes. The caller must be done reading records out of it: the -// record readers decode into freshly allocated values (encoding/json copies -// every string, json.Number included), so "done" means the last Next has -// returned — nothing a handed-back record holds points into these bytes. +// maxPooledBufferBytes. The caller must be done with the request: the body +// goes to the type layer as-is, and what is published is the rows the type +// layer exported into its own buffer — nothing published points into these +// bytes. func putBodyBuffer(buf *bytes.Buffer) { if buf == nil || buf.Cap() > maxPooledBufferBytes { return diff --git a/internal/api/record_reader.go b/internal/api/content_type.go similarity index 59% rename from internal/api/record_reader.go rename to internal/api/content_type.go index cef51468..85c5a614 100644 --- a/internal/api/record_reader.go +++ b/internal/api/content_type.go @@ -1,54 +1,19 @@ package api import ( - "bufio" - "bytes" - "encoding/json" "errors" "fmt" - "io" "mime" "slices" "strings" -) - -const ( - // maxNDJSONLineBytes caps a single NDJSON record so one pathological line - // can't force an unbounded read buffer. 10 MiB is far above any realistic - // flat ingest record; a line larger than this aborts the whole request. - maxNDJSONLineBytes = 10 << 20 // 10 MiB - // maxSniffBytes bounds how far the arity peek looks for the first - // non-whitespace byte. Far beyond any reasonable amount of leading - // whitespace; a body that is only whitespace within this window is treated - // as empty. - maxSniffBytes = 512 + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) -// recordReader yields ingest records one at a time from a request body. Each -// concrete reader covers one wire format (single JSON object, JSON array, -// NDJSON, and — later — CSV), so the handler stays format-agnostic and new -// formats / transports (streaming uploads) slot in behind this one interface. -// -// Next returns io.EOF at the clean end of the body. A *recordSyntaxError is a -// recoverable per-record decode failure — the framing let the reader resync, so -// the handler records it and continues. Any other non-EOF error is fatal to the -// request. -type recordReader interface { - Next() (map[string]any, error) -} - -// recordSyntaxError marks a per-record decode failure the reader recovered from -// (the framing let it skip to the next record). The batch handler turns it into -// a recordResult error; it carries no HTTP status because the decode layer sits -// below the validation/permission layer that owns status codes. -type recordSyntaxError struct{ msg string } - -func (e *recordSyntaxError) Error() string { return e.msg } - -// errEmptyBody is returned by newRecordReader when the body has no content (no -// non-whitespace byte within the sniff window). The handler maps it to a 400. -var errEmptyBody = errors.New("empty body") +// maxSniffBytes bounds how far the arity peek looks for the first +// non-whitespace byte. Far beyond any reasonable amount of leading whitespace; +// a body that is only whitespace within this window is treated as empty. +const maxSniffBytes = 512 // errUnsupportedContentType is returned when the request declares no // Content-Type, one whose media type is not in the accepted list, several that @@ -69,25 +34,47 @@ var errConflictingContentType = errors.New("conflicting content type") // IngestFormat is the wire format of an ingest request body. It comes from the // request's declared Content-Type and nothing else: the body never overrides -// what the client said it sent, so a caller can always tell how their bytes -// will be read without knowing what the first one happens to be. +// what the client said it sent, so a caller can always tell how their bytes will +// be read without knowing what the first one happens to be. +// +// It is NOT the same thing as the chtypes format the parser is handed — see +// wire(). The two JSON families are one format to ClickHouse and two formats +// here, because they differ in how a body frames its records, which is what +// decides the response shape and how many records the request contains. type IngestFormat int const ( // FormatJSON is the application/json family: one flat object, or a // top-level array of them. Which of the two is the body's own business — // the first non-whitespace byte picks it — because both are the same - // format, differing only in arity. + // format to the parser, differing only in arity. FormatJSON IngestFormat = iota - // FormatNDJSON is newline-delimited JSON: one flat object per line. A line - // that is not a JSON object is a per-record error, never a reason to - // re-read the body as something else. + // FormatNDJSON is newline-delimited JSON: one flat object per line. Byte + // for byte the same format ClickHouse calls JSONEachRow; the distinction + // from FormatJSON survives only so a declared-NDJSON body is always a + // batch and never has its arity sniffed. FormatNDJSON - // FormatCSV plugs in here once ingest reads CSV: add the media types to - // acceptedContentTypes, which advertises them in the 415 automatically, and - // the reader to newRecordReader. Do NOT add them in ingestFormatOne — it - // resolves by scanning that one table, and a second list is the drift this - // arrangement exists to prevent. + // FormatCSV is bare `text/csv`: positional in the table's wire columns + // (declaration order minus MATERIALIZED, ALIAS and EPHEMERAL), under + // ClickHouse's own header auto-detection — a first line that spells the + // column names is consumed as a header, not a record. + FormatCSV + // FormatTSV is FormatCSV's tab-separated twin. + FormatTSV + // FormatCSVWithNames is CSV whose first line names the columns, in any + // order — `text/csv; header=present`, RFC 4180 §3's parameter. The header + // is not a record; a column it omits takes its DEFAULT, and a name the + // table (or the role's projection of it) lacks refuses the request with + // ClickHouse's code 117. + FormatCSVWithNames + // FormatTSVWithNames is FormatCSVWithNames' tab-separated twin. + FormatTSVWithNames + // FormatCSVPositional is `text/csv; header=absent`: strictly positional, + // detection off, so a header line is one record that fails to parse with + // ClickHouse's code 27. + FormatCSVPositional + // FormatTSVPositional is FormatCSVPositional's tab-separated twin. + FormatTSVPositional ) // String renders a format for error messages and logs. @@ -97,11 +84,54 @@ func (f IngestFormat) String() string { return "json" case FormatNDJSON: return "ndjson" + case FormatCSV: + return "csv" + case FormatTSV: + return "tsv" + case FormatCSVWithNames: + return "csvwithnames" + case FormatTSVWithNames: + return "tsvwithnames" + case FormatCSVPositional: + return "csv" + case FormatTSVPositional: + return "tsv" default: return "unknown" } } +// wire is the format chtypes parses the body as. Both JSON families collapse +// onto JSONEachRow: an NDJSON body, a bare object, concatenated objects and a +// newline-framed array are all the same input to ClickHouse's own reader. +func (f IngestFormat) wire() typelayer.Format { + switch f { + case FormatCSV, FormatCSVPositional: + return typelayer.FormatCSV + case FormatTSV, FormatTSVPositional: + return typelayer.FormatTSV + case FormatCSVWithNames: + return typelayer.FormatCSVWithNames + case FormatTSVWithNames: + return typelayer.FormatTSVWithNames + case FormatJSON, FormatNDJSON: + return typelayer.FormatJSONEachRow + default: + return typelayer.FormatJSONEachRow + } +} + +// options are the parse options the format adds to wire(): header detection is +// off exactly for the header=absent pair. +func (f IngestFormat) options() typelayer.IngestOptions { + return typelayer.IngestOptions{StrictPositional: f == FormatCSVPositional || f == FormatTSVPositional} +} + +// alwaysBatch reports whether this format's response is the per-record batch +// shape whatever the body holds. Only the JSON family has an arity question, +// and the first non-whitespace byte answers it (see firstNonSpace). +func (f IngestFormat) alwaysBatch() bool { return f != FormatJSON } + // acceptedContentTypes maps every media type ingest reads to the format it // selects, in the order the 415 body advertises them. The first entry of each // family is the canonical spelling — the TS SDK sends those two. It is the @@ -113,15 +143,31 @@ func (f IngestFormat) String() string { // architecture.md all failed to name — and no test can close that direction by // enumeration, because the complement is unbounded. One table closes it by // construction. +// +// header is the `header` parameter an entry requires: "" matches a declaration +// without one; "present" and "absent" match only that value. It is the one +// parameter that decides a format, and only for the two media types that list +// non-empty entries. RFC 4180 §3 defines it for text/csv, and it maps onto +// ClickHouse's own behaviour three ways: present is the WithNames format, +// absent is strictly positional (detection off), and none leaves ClickHouse's +// default header auto-detection on. text/tab-separated-values takes the same +// mapping (IANA defines no parameters for it). var acceptedContentTypes = []struct { mediaType string + header string format IngestFormat }{ - {"application/json", FormatJSON}, - {"application/x-ndjson", FormatNDJSON}, - {"application/ndjson", FormatNDJSON}, - {"application/jsonl", FormatNDJSON}, - {"application/jsonlines", FormatNDJSON}, + {"application/json", "", FormatJSON}, + {"application/x-ndjson", "", FormatNDJSON}, + {"application/ndjson", "", FormatNDJSON}, + {"application/jsonl", "", FormatNDJSON}, + {"application/jsonlines", "", FormatNDJSON}, + {"text/csv", "", FormatCSV}, + {"text/csv", "present", FormatCSVWithNames}, + {"text/csv", "absent", FormatCSVPositional}, + {"text/tab-separated-values", "", FormatTSV}, + {"text/tab-separated-values", "present", FormatTSVWithNames}, + {"text/tab-separated-values", "absent", FormatTSVPositional}, } // supportedContentTypes is what the 415 body lists and the docs quote, derived @@ -130,134 +176,13 @@ var supportedContentTypes = func() []string { out := make([]string, len(acceptedContentTypes)) for i, a := range acceptedContentTypes { out[i] = a.mediaType + if a.header != "" { + out[i] += "; header=" + a.header + } } return out }() -// errUnterminatedArray marks a JSON array body that ended before its closing -// ']' (a truncated / cut-off upload). It is deliberately NOT io.EOF so the -// batch loop fails the whole request (400) instead of treating the records that -// did arrive as a complete, successful batch. -var errUnterminatedArray = errors.New("unterminated json array") - -// objectReader decodes exactly one flat JSON object — the single-object ingest -// path. A second Next returns io.EOF. Trailing bytes after the first object are -// ignored (matching the historical single-object behavior), so the response -// shape never depends on what follows the object. -type objectReader struct { - dec *json.Decoder - done bool -} - -func (o *objectReader) Next() (map[string]any, error) { - if o.done { - return nil, io.EOF - } - o.done = true - var m map[string]any - if err := o.dec.Decode(&m); err != nil { - return nil, err // fatal: handler maps this to 400 invalid json - } - return m, nil -} - -// arrayReader streams the elements of a top-level JSON array. A wrong-typed -// element (a scalar/array where an object was expected) yields a -// *json.UnmarshalTypeError, which leaves the decoder in sync — recoverable, so -// it becomes a per-record error and iteration continues. A *json.SyntaxError -// desyncs the decoder and is returned as fatal. -type arrayReader struct { - dec *json.Decoder - started bool - done bool -} - -func (a *arrayReader) Next() (map[string]any, error) { - if a.done { - return nil, io.EOF - } - if !a.started { - if _, err := a.dec.Token(); err != nil { // consume the opening '[' - a.done = true - return nil, err - } - a.started = true - } - if !a.dec.More() { - a.done = true - // More() reports false not only at a clean ']', but also on a read error - // and on a truncated array (EOF before ']'), swallowing both — which - // would let dropped records masquerade as a complete partial-200 insert. - // Read the closing token to tell the cases apart: only a ']' ends the - // batch; a missing or non-']' close means the upload was cut off (→ 400, - // via errUnterminatedArray, which is NOT io.EOF) and fails the whole - // request. - // - // The body-cap case no longer reaches here — the handler buffers the - // whole body first, so a cap overflow is a 413 before any reader exists, - // and this operates on an in-memory slice that cannot fail a read. - tok, err := a.dec.Token() - if err != nil { - if errors.Is(err, io.EOF) { - return nil, errUnterminatedArray - } - return nil, err - } - if d, ok := tok.(json.Delim); !ok || d != ']' { - return nil, errUnterminatedArray - } - return nil, io.EOF - } - var m map[string]any - if err := a.dec.Decode(&m); err != nil { - if _, ok := errors.AsType[*json.UnmarshalTypeError](err); ok { - // Decoder stayed in sync past the bad element — recoverable. - return nil, &recordSyntaxError{"record must be a JSON object"} - } - // Syntax/read error: the decoder is desynced, the rest of the array is - // unrecoverable. Don't try to read the closing ']'. - a.done = true - if errors.Is(err, io.EOF) { - // More() said an element followed (e.g. after a trailing comma) but - // the stream ended — a truncated array, not a clean close. Map to a - // fatal error rather than the io.EOF the batch loop treats as "done". - return nil, errUnterminatedArray - } - return nil, err - } - return m, nil -} - -// lineReader yields one record per non-blank line of an NDJSON body. It recovers -// from both type and syntax errors per line (the newline reframes the next -// record), so a single malformed line never blocks the rest of the batch. -type lineReader struct { - sc *bufio.Scanner -} - -func (l *lineReader) Next() (map[string]any, error) { - for l.sc.Scan() { - line := bytes.TrimSpace(l.sc.Bytes()) - if len(line) == 0 { - continue // skip blank lines between records - } - var m map[string]any - dec := json.NewDecoder(bytes.NewReader(line)) - dec.UseNumber() - if err := dec.Decode(&m); err != nil { - return nil, &recordSyntaxError{"invalid json"} - } - return m, nil - } - if err := l.sc.Err(); err != nil { - // A line exceeding maxNDJSONLineBytes (bufio.ErrTooLong) — the scanner - // can't resume, so fail the request. Not the body cap: that trips in the - // handler before this reader is built. - return nil, err - } - return nil, io.EOF -} - // resolveContentType resolves the Content-Type header set to the format ingest // reads the body as. Content-Type is a singleton field (RFC 9110 §8.3) and §5.3 // forbids repeating it, so a duplicate is malformed however it is spelled. §8.3 @@ -283,8 +208,10 @@ func resolveContentType(values []string) (IngestFormat, int, error) { return f, -1, err } -// ingestFormatOne resolves ONE header line, parsed per RFC 9110 §8.3. Only the -// media type decides the format; no malformed parameter costs the request. +// ingestFormatOne resolves ONE header line, parsed per RFC 9110 §8.3. The media +// type decides the format, plus the `header` parameter for the two CSV-family +// types (see acceptedContentTypes); no other malformed parameter costs the +// request. // // That rule needs two steps, because Go splits parse failures in a way the rule // does not. ErrInvalidMediaParameter leaves the media type parsed and returned, @@ -299,8 +226,13 @@ func resolveContentType(values []string) (IngestFormat, int, error) { // member there reads an NDJSON body as one object, dropping every record past // it behind a 200. The error cannot distinguish that from a comma inside data, // so such a line is refused rather than guessed at (#563). +// +// The same caution applies to `header`: on a line whose parameters did not +// parse, a header parameter cannot be read, and guessing "none" would ingest +// a declared header line as data. Such a line is refused when it mentions one; +// a header value other than present/absent is refused too. func ingestFormatOne(v string) (IngestFormat, error) { - mediaType, _, err := mime.ParseMediaType(v) + mediaType, params, err := mime.ParseMediaType(v) if err != nil { if strings.ContainsRune(v, ',') { return FormatJSON, errUnsupportedContentType @@ -313,14 +245,38 @@ func ingestFormatOne(v string) (IngestFormat, error) { mediaType = base } } + header := "" + if headerDecides(mediaType) { + if err != nil && strings.Contains(strings.ToLower(v), "header") { + return FormatJSON, errUnsupportedContentType + } + switch strings.ToLower(params["header"]) { + case "": + case "present", "absent": + header = strings.ToLower(params["header"]) + default: + return FormatJSON, errUnsupportedContentType + } + } for _, a := range acceptedContentTypes { - if a.mediaType == mediaType { + if a.mediaType == mediaType && a.header == header { return a.format, nil } } return FormatJSON, errUnsupportedContentType } +// headerDecides reports whether the `header` parameter selects the format for +// this media type — true only for a type with header-qualified entries. +func headerDecides(mediaType string) bool { + for _, a := range acceptedContentTypes { + if a.mediaType == mediaType && a.header != "" { + return true + } + } + return false +} + // mediaTypePrefix is everything before the first ";" — the media type without // its parameters. Only ingestFormatOne's re-parse uses it, and only on a line // with no comma, so it cannot resurrect a joined declaration. @@ -329,40 +285,6 @@ func mediaTypePrefix(v string) string { return base } -// newRecordReader picks a reader for an already-resolved format over an -// already-buffered body, looking at the bytes only to choose arity within the -// JSON family ('[' → array, else → single object). The declared format is -// authoritative: an NDJSON body is read as NDJSON whatever its first byte, so a -// line that isn't a JSON object fails as a per-record error rather than silently -// re-framing the whole request. batch is false only for the single-object path; -// true for array/NDJSON. -// -// It takes the format rather than a Content-Type on purpose: resolving here as -// well as in the handler would put one rule in two places, which is how the -// joined and repeated paths came to disagree before. The only error this can -// return is errEmptyBody. The caller resolves the format with resolveContentType -// (which owns the 415) and is expected to have bounded the body via -// http.MaxBytesReader before buffering it. -func newRecordReader(format IngestFormat, body []byte) (rr recordReader, batch bool, err error) { - first, ok := firstNonSpace(body) - if !ok { - return nil, false, errEmptyBody - } - - if format == FormatNDJSON { - sc := bufio.NewScanner(bytes.NewReader(body)) - sc.Buffer(make([]byte, 0, 64*1024), maxNDJSONLineBytes) - return &lineReader{sc: sc}, true, nil - } - - dec := json.NewDecoder(bytes.NewReader(body)) - dec.UseNumber() - if first == '[' { - return &arrayReader{dec: dec}, true, nil - } - return &objectReader{dec: dec}, false, nil -} - // firstNonSpace returns the body's first non-whitespace byte. ok is false when // there is none within the sniff window — the same bound the streaming reader // used, kept so a body of leading whitespace longer than the window still reads @@ -382,10 +304,10 @@ func firstNonSpace(body []byte) (byte, bool) { // emptyBodyMessage tailors the empty-body 400 message to the declared format so // an NDJSON caller still gets the familiar "empty ndjson body". func emptyBodyMessage(format IngestFormat) string { - if format == FormatNDJSON { - return "empty ndjson body" + if format == FormatJSON { + return "empty body" } - return "empty body" + return "empty " + format.String() + " body" } // Bounds on what a caller-supplied Content-Type may cost us when echoed back. diff --git a/internal/api/errors.go b/internal/api/errors.go index b753668a..edd6ebbe 100644 --- a/internal/api/errors.go +++ b/internal/api/errors.go @@ -23,10 +23,17 @@ func writeJSONError(w http.ResponseWriter, status int, message string) { // errorBody is the error envelope. Code and Retryable are set where the // handler knows them (writeCHError); a bare {"error": …} otherwise. +// +// ExceptionCode is ClickHouse's own numeric error code, set only where +// ClickHouse's parser is what refused (ingest: 117 unknown field, 27 +// unparseable value, …). Zero is omitted, so its presence always means +// ClickHouse answered — a gateway rejection never carries one. Code stays the +// string class. type errorBody struct { - Error string `json:"error"` - Code string `json:"code,omitempty"` - Retryable *bool `json:"retryable,omitempty"` + Error string `json:"error"` + Code string `json:"code,omitempty"` + ExceptionCode int `json:"exception_code,omitempty"` + Retryable *bool `json:"retryable,omitempty"` } func writeJSONErrorBody(w http.ResponseWriter, status int, body errorBody) { diff --git a/internal/api/ingest.go b/internal/api/ingest.go index fa1ea816..28d6277f 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -5,13 +5,16 @@ import ( "encoding/json" "errors" "fmt" - "io" "log/slog" + "maps" "math" "net/http" + "reflect" + "slices" "sort" "strconv" "strings" + "sync" "time" "github.com/Wave-RF/WaveHouse/internal/auth" @@ -21,6 +24,8 @@ import ( "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" "go.opentelemetry.io/otel" "go.opentelemetry.io/otel/attribute" @@ -39,13 +44,20 @@ const maxReportedResults = 10000 // ingestWindow is how many records a batch prepares before reserving, // publishing and committing them together: one dedupe call per phase per -// window rather than per record, and at most one window of encoded rows held. +// window rather than per record, and at most one window of encoded envelopes +// held. The exported rows they are built from are held for the whole request — +// one parse per body — and are bounded by the body cap. const ingestWindow = 256 // IngestHandler handles POST /v1/ingest?table={table} type IngestHandler struct { // Registry yields the request tenant's schema registry. Registry RegistrySource + // Types is the process's type layer: ClickHouse's own parser, which judges + // every record against the request tenant's compiled schema. Nil is never + // "skip validation" — nothing else on this path looks at a value — so a + // handler without one refuses every request it would have judged with a 503. + Types *typelayer.Engine // Dedup resolves the request tenant's deduplicator — the tenant's own // store, picked off the store the handler already holds (#583 story 7; // dedupe.Stores in production). nil when no dedupe store is wired (tests). @@ -61,12 +73,6 @@ type IngestHandler struct { Publisher mq.Publisher PolicySource PolicySource - // Validator and Checker are the per-record seams a native type layer will - // take over (see ingest_seams.go). Both are optional: nil means the default - // implementation, which is today's behavior unchanged. - Validator RecordValidator - Checker InsertChecker - // maxRequestBytes optionally overrides the default inbound request body cap // (maxRequestBodyBytes). When 0, the default applies. Exists so same-package // tests can pin the cap-overflow path without allocating 16 MiB per run; not @@ -74,6 +80,13 @@ type IngestHandler struct { maxRequestBytes int64 // window overrides ingestWindow when > 0, for tests and benchmarks. window int + + // noticeMu guards noticeLast, the last time each rate-limited notice was + // logged. A tenant the type layer cannot serve, or a policy whose injected + // literal will not compile, is a standing condition: one line per request + // would bury the rest of the log under it. + noticeMu sync.Mutex + noticeLast map[string]time.Time } func NewIngestHandler(registry RegistrySource, pub mq.Publisher) *IngestHandler { @@ -101,14 +114,14 @@ var dedupeDisabledCounter, _ = otel.Meter("wavehouse-ingest").Int64Counter( metric.WithDescription("Ingested records published without dedupe because the store was switched off while settings said enabled (reload window)"), ) -// batchResult is the response body for any multi-record ingest (a JSON array, -// an NDJSON batch, and — later — CSV). The status is 200 whenever the body was -// readable and the records were processed; per-record rejections (malformed -// JSON, schema or permission failures) are reported in Results without failing -// the whole request, so one bad record never obscures the rest of the batch -// (issue #195). Whole-request conditions abort with a non-200 status instead — -// see requestAbort for the list, which lives there and only there, because -// stating it in three places is how two of them went stale. +// batchResult is the response body for any multi-record ingest: a JSON array, +// an NDJSON batch, a CSV or a TSV body. The status is 200 whenever the body was +// readable and the records were processed; per-record rejections (unparseable +// values, unknown columns, failed check clauses) are reported in Results without +// failing the whole request, so one bad record never obscures the rest of the +// batch (issue #195). Whole-request conditions abort with a non-200 status +// instead — see requestAbort for the list, which lives there and only there, +// because stating it in three places is how two of them went stale. type batchResult struct { Total int `json:"total"` // records read from the body Succeeded int `json:"succeeded"` // records validated + published @@ -126,15 +139,22 @@ type recordResult struct { Ok bool `json:"ok,omitempty"` Duplicate bool `json:"duplicate,omitempty"` Error string `json:"error,omitempty"` + // ExceptionCode is ClickHouse's own error code when the record was refused + // by the server's parser (117 unknown field — which includes a column the + // role may not write, 27 unparseable value, 6 out of range). Absent for a + // gateway rejection — a failed check clause is our verdict, not + // ClickHouse's, and must not be dressed as one. + ExceptionCode int `json:"exception_code,omitempty"` } -// recordReject is a per-record rejection: this record is bad (failed schema -// validation or a column/check permission rule), but the rest of the batch can -// still proceed. The single-object path maps Status to the HTTP code; the batch -// path records Message against the index and keeps going. +// recordReject is a per-record rejection: this record is bad (ClickHouse's +// parser refused it, or a policy check clause did), but the rest of the batch +// can still proceed. The single-object path maps Status to the HTTP code; the +// batch path records Message against the index and keeps going. type recordReject struct { - Status int - Message string + Status int + Message string + ExceptionCode int // ClickHouse's code; 0 for a gateway rejection (see recordResult) } // requestAbort is a whole-request failure: this record and every one that @@ -147,17 +167,50 @@ type recordReject struct { // makes the batch safe to retry: publish backpressure (503), an unreachable // broker (503, mq.ErrUnavailable), a publish/marshal failure (500), a dedupe // store that cannot answer (503) or fails (500), an id another request holds -// (503). +// (503), a type layer that cannot judge the tenant's table (503). // -// One is not. An insert grant that resolved for the other operation is a 403 and -// a caller/config bug — retrying cannot help. It aborts rather than rejecting -// per record because the grant is resolved ONCE per request, so it is true for -// every record or none; as a per-record reject a 10k batch would report 10k -// independent permission failures for a single mis-wired grant. +// Two are not, and retrying either unchanged cannot help: +// - An insert grant that resolved for the other operation is a 403 and a +// caller/config bug. It aborts rather than rejecting per record because the +// grant is resolved ONCE per request, so it is true for every record or +// none; as a per-record reject a 10k batch would report 10k independent +// permission failures for a single mis-wired grant. +// - A header-format body whose header ClickHouse refuses — a name the table +// or the role lacks, or a name given twice — is a 400 with the +// clickhouse.rejected class and ClickHouse's own code (117). The header is +// not a record, and no record was read past it. type requestAbort struct { - Status int - Message string - RetryAfter string // non-empty → emit a Retry-After header + Status int + Message string + Code string // the error class, when one applies (codeCHRejected) + ExceptionCode int // ClickHouse's code, when its parser refused the body as a whole + RetryAfter string // non-empty → emit a Retry-After header +} + +// ingestRun is one request after its body has been ruled on: chtypes' verdict +// per record, its check answer included, and the wire columns the accepted rows +// were exported with. Both response shapes read their records out of it, so +// the single-object and batch paths cannot disagree about what a record's +// outcome is — only about how it is rendered. +type ingestRun struct { + store *settings.Store + table, scope string + now time.Time + batch typelayer.Batch + // wire is the role's wire columns, copied off the handle that exported the + // rows: the envelope's Columns, and where the dedupe id sits in a row. + wire []string + // records is how many records the request contains: the body's own framing + // before chtypes has read it (an empty array is zero, anything else is at + // least one), then len(batch.Rows) once it has answered. + records int + // checkColumns names the check clauses a record's check answer came from, + // for the rejection message. The filter is AND-joined over all of them, so + // a false verdict does not say which one failed — with one clause it does. + checkColumns []string + // checkGuard is the rejection every otherwise-acceptable record gets when + // the role's check clauses name columns no record can carry. + checkGuard *recordReject } func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { @@ -181,8 +234,6 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { r = r.WithContext(ctx) - // TODO: what should the order of these be to maximize speed + limit risk of data leakage or DoS/resource exhaustion? - if table == "" { slog.ErrorContext(ctx, "missing table parameter in request") writeJSONError(w, http.StatusBadRequest, "missing table") @@ -262,18 +313,27 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { return } - // Bound the inbound body (parity with /v1/ops/query; also caps the - // array/stream decode vectors). See query.go for maxRequestBodyBytes. + // A tenant the type layer cannot judge is refused before the body is read, + // like a schema not discovered yet: nothing in the body can change the + // answer, so there is no reason to buffer up to 16 MiB of it. The handle is + // not held across the read — a slow upload must not hold off a rebind. + if abort := h.typesReady(ctx, store.Tenant(), table); abort != nil { + writeAbort(w, abort) + return + } + + // Bound the inbound body (parity with /v1/ops/query). See query.go for + // maxRequestBodyBytes. reqCap := int64(maxRequestBodyBytes) if h.maxRequestBytes > 0 { reqCap = h.maxRequestBytes } r.Body = http.MaxBytesReader(w, r.Body, reqCap) - // Read the whole (already-capped) body up front and run the readers over - // those bytes rather than the live connection. The buffer comes from a pool - // and goes back on the way out — every record the readers hand back is - // freshly allocated, so nothing downstream points into it. + // Read the whole (already-capped) body up front and hand those bytes to + // ClickHouse's own parser. The buffer comes from a pool and goes back on the + // way out; what is published is the type layer's own export of the records, + // a separate buffer, so nothing downstream points into this one. body := getBodyBuffer() defer putBodyBuffer(body) if _, err := body.ReadFrom(r.Body); err != nil { @@ -285,57 +345,128 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { return } - // The reader is built from the RESOLVED format, not from a second look at - // the header. Re-resolving here would duplicate the rule in two places, - // which is how the joined and repeated paths came to disagree before. The - // only error left is an empty body. - rr, batch, err := newRecordReader(format, body.Bytes()) - if err != nil { + // Framing: the declared format says how the body frames its records, and the + // first non-whitespace byte answers the one question left inside the JSON + // family — array or single object. That byte is the ONLY thing the body gets + // to decide, and it decides the response shape, never the format. + first, ok := firstNonSpace(body.Bytes()) + if !ok { slog.ErrorContext(ctx, "empty ingest body", "table", table, "format", format.String()) writeJSONError(w, http.StatusBadRequest, emptyBodyMessage(format)) return } + batchShape := format.alwaysBatch() || first == '[' + records := 1 + if format == FormatJSON && first == '[' { + n, framed := reframeArray(body.Bytes()) + if !framed { + // Brackets that do not balance: a truncated upload or a structural + // syntax error. Neither can be salvaged per record, and reporting the + // records that did arrive as a complete batch is the failure this + // refusal exists to prevent. + slog.WarnContext(ctx, "ingest read error", "error", "unterminated json array", "table", table) + writeJSONError(w, http.StatusBadRequest, "invalid json: unterminated json array") + return + } + records = n + } + // Otherwise a single-object body is one record (concatenated objects after + // it are ignored, as they always have been — declare NDJSON to batch them, + // #561), and a line-framed body has at least the record its first byte + // starts. The real count is chtypes' own, taken once it has answered. - if batch { - h.handleBatch(ctx, w, rr, reqCap, store, table, scope, schema, perms, role, now, h.policyCheckGuard(ctx, table, role, schema, perms)) + guard := h.policyCheckGuard(ctx, table, role, schema, perms) + shape, preds, checkColumns, abort := h.insertShape(ctx, table, role, schema, perms, guard) + if abort != nil { + writeAbort(w, abort) return } - h.handleSingle(ctx, w, rr, reqCap, store, table, scope, schema, perms, role, now, h.policyCheckGuard(ctx, table, role, schema, perms)) -} -// handleSingle ingests a lone flat JSON object and preserves the GA response -// contract: 200 {"ok":true} (or {"duplicate":true} when dedup skips it), or the -// matching non-200 on validation / permission / whole-request failure. -func (h *IngestHandler) handleSingle( - ctx context.Context, - w http.ResponseWriter, - rr recordReader, - reqCap int64, - store *settings.Store, - table, scope string, - schema *discovery.TableSchema, - perms *policy.ResolvedPermissions, - role string, - now time.Time, - checkGuard *recordReject, -) { - data, err := rr.Next() - if err != nil { - // Unreachable while the body is buffered — a bytes.Reader cannot produce - // a *http.MaxBytesError, and the cap is enforced at body.ReadFrom. Kept - // because it is the correct mapping if a reader ever streams again. - if writeMaxBytesError(w, err, reqCap) { + run := &ingestRun{ + store: store, table: table, scope: scope, now: now, + records: records, checkColumns: checkColumns, checkGuard: guard, + } + if records > 0 { + if abort := h.judge(ctx, run, shape, format, body.Bytes(), preds); abort != nil { + writeAbort(w, abort) return } - slog.ErrorContext(ctx, "invalid json payload", "error", err, "table", table) - writeJSONError(w, http.StatusBadRequest, "invalid json") + } + + if batchShape { + h.writeBatch(ctx, w, run) return } + h.writeSingle(ctx, w, run) +} + +// typesReady answers whether the type layer can judge table for tenant id now, +// with the 503 every record of the request would get when it cannot. +func (h *IngestHandler) typesReady(ctx context.Context, id tenant.ID, table string) *requestAbort { + if h.Types == nil { + // Nothing else on this path inspects a value, so an unwired type layer + // cannot mean "accept anything" — the same fail-closed direction the + // stream's row evaluator takes. + h.logUnavailable(ctx, id, table, "no type layer is wired") + return unavailableAbort() + } + tbl, err := h.Types.Table(id, table) + if err != nil { + return h.typesFailed(ctx, id, table, err) + } + tbl.Release() + return nil +} + +// judge asks ClickHouse's own parser for a verdict per record — ONE call for +// the whole body, no chunking — with the role's insert check clauses, if any, +// compiled to ONE filter and answered by that same parse. +// +// The role's projection of the table answers column policy and the injected +// check values inside the parser, so no Go code walks the record's keys. It is +// held for the parse alone: the verdicts and exported rows are the run's own +// once the call returns, so dedupe and publish latency never hold off a rebind. +// +// A Go-level failure is an unavailable handle (the type layer's outage) or +// ours, and neither is the caller's record to fix. A body ClickHouse refused as +// a whole is the caller's to fix, and is a 400 with its code. +func (h *IngestHandler) judge(ctx context.Context, run *ingestRun, shape typelayer.RoleShape, format IngestFormat, body []byte, preds []policy.Predicate) *requestAbort { + id := run.store.Tenant() + tbl, abort := h.roleTable(ctx, id, run.table, shape) + if abort != nil { + return abort + } + batch, err := tbl.IngestWith(format.wire(), format.options(), body, preds...) + run.wire = slices.Clone(tbl.WireColumns) + tbl.Release() + if err != nil { + if typelayer.IsUnavailable(err) { + return h.typesFailed(ctx, id, run.table, err) + } + slog.ErrorContext(ctx, "record validation failed", "error", err, "table", run.table) + return &requestAbort{Status: http.StatusInternalServerError, Message: "validation failed"} + } + if r := batch.Refused; r != nil { + slog.WarnContext(ctx, "ingest body refused by the parser", "error", r.Message, "exception_code", r.Code, "table", run.table) + return &requestAbort{Status: http.StatusBadRequest, Message: r.Message, Code: codeCHRejected, ExceptionCode: r.Code} + } + run.batch = batch + // chtypes' per-record answer is the record count: a JSON array sent as + // NDJSON is however many elements its reader took, a blank line is nothing. + // When it gave no per-record detail the padded batch is still index-shaped, + // so the same rule keeps every later index in range. + run.records = len(batch.Rows) + return nil +} - rec, abort := h.prepareRecord(ctx, store, table, scope, schema, perms, role, data, now, checkGuard) +// writeSingle answers a lone flat JSON object and preserves the GA response +// contract: 200 {"ok":true} (or {"duplicate":true} when dedup skips it), or the +// matching non-200 on refusal / permission / whole-request failure. +func (h *IngestHandler) writeSingle(ctx context.Context, w http.ResponseWriter, run *ingestRun) { + rec, abort := h.prepareVerdict(ctx, run, 0) if abort == nil && rec.reject == nil { window := []pendingRecord{rec} - abort = h.ingestWindow(ctx, store, table, scope, window) + abort = h.ingestWindow(ctx, run.store, run.table, run.scope, window) rec = window[0] } if abort != nil { @@ -343,7 +474,7 @@ func (h *IngestHandler) handleSingle( return } if rec.reject != nil { - writeJSONError(w, rec.reject.Status, rec.reject.Message) + writeJSONErrorBody(w, rec.reject.Status, errorBody{Error: rec.reject.Message, ExceptionCode: rec.reject.ExceptionCode}) return } if rec.duplicate { @@ -352,32 +483,19 @@ func (h *IngestHandler) handleSingle( return } - slog.InfoContext(ctx, "event successfully ingested", "table", table) + slog.InfoContext(ctx, "event successfully ingested", "table", run.table) w.Header().Set("Content-Type", "application/json") _ = json.NewEncoder(w).Encode(map[string]bool{"ok": true}) } -// handleBatch ingests a multi-record body (JSON array or NDJSON), running each -// record through the same validate → authorize → dedup → publish pipeline as a -// single insert. A record that fails validation or a PER-RECORD permission rule -// (a denied column, a failed check clause) — or that the reader couldn't decode -// — is recorded against its index and the batch continues; a whole-request -// condition aborts it (see requestAbort). Returns 200 with a per-record summary -// once the body is consumed. -func (h *IngestHandler) handleBatch( - ctx context.Context, - w http.ResponseWriter, - rr recordReader, - reqCap int64, - store *settings.Store, - table, scope string, - schema *discovery.TableSchema, - perms *policy.ResolvedPermissions, - role string, - now time.Time, - checkGuard *recordReject, -) { - result := batchResult{Results: []recordResult{}} +// writeBatch answers a multi-record body (JSON array, NDJSON, CSV or TSV), +// running each verdict through the same prepare → reserve → publish → commit +// pipeline as a single insert, a window at a time. A record ClickHouse's +// parser refused, or one a check clause denied, is recorded against its index +// and the batch continues; a whole-request condition aborts it (see +// requestAbort). Returns 200 with a per-record summary. +func (h *IngestHandler) writeBatch(ctx context.Context, w http.ResponseWriter, run *ingestRun) { + result := batchResult{Total: run.records, Results: []recordResult{}} size := h.window if size <= 0 { size = ingestWindow @@ -386,7 +504,7 @@ func (h *IngestHandler) handleBatch( // flush runs the window's records through reserve → publish → commit and // reports them in order; false when it aborted the request. flush := func() bool { - if abort := h.ingestWindow(ctx, store, table, scope, window); abort != nil { + if abort := h.ingestWindow(ctx, run.store, run.table, run.scope, window); abort != nil { writeAbort(w, abort) return false } @@ -397,37 +515,8 @@ func (h *IngestHandler) handleBatch( return true } - for { - data, err := rr.Next() - if errors.Is(err, io.EOF) { - break - } - if err != nil { - if rse, ok := errors.AsType[*recordSyntaxError](err); ok { - result.Total++ - window = append(window, pendingRecord{index: result.Total, reject: &recordReject{Message: rse.Error()}}) - if len(window) == size && !flush() { - return - } - continue - } - // Unreachable while the body is buffered — a bytes.Reader cannot produce - // a *http.MaxBytesError, and the cap is enforced at body.ReadFrom. Kept - // because it is the correct mapping if a reader ever streams again. - if writeMaxBytesError(w, err, reqCap) { - return - } - // A fatal stream error (JSON array syntax error, or an oversized - // NDJSON line) — the reader can't resume, so fail the request rather - // than report a misleading partial summary. Not a body read error: - // the readers run over an in-memory slice now. - slog.WarnContext(ctx, "ingest read error", "error", err, "table", table) - writeJSONError(w, http.StatusBadRequest, "invalid json: "+err.Error()) - return - } - - result.Total++ - rec, abort := h.prepareRecord(ctx, store, table, scope, schema, perms, role, data, now, checkGuard) + for i := range run.records { + rec, abort := h.prepareVerdict(ctx, run, i) if abort != nil { // Whole-request failure: surface the status rather than recording a // request-scoped condition as per-record loss (see requestAbort). @@ -435,7 +524,7 @@ func (h *IngestHandler) handleBatch( writeAbort(w, abort) return } - rec.index = result.Total + rec.index = i + 1 window = append(window, rec) if len(window) == size && !flush() { return @@ -445,7 +534,7 @@ func (h *IngestHandler) handleBatch( return } - slog.InfoContext(ctx, "batch ingested", "table", table, + slog.InfoContext(ctx, "batch ingested", "table", run.table, "total", result.Total, "succeeded", result.Succeeded, "failed", result.Failed, "duplicates", result.Duplicates) w.Header().Set("Content-Type", "application/json") @@ -453,14 +542,14 @@ func (h *IngestHandler) handleBatch( } // add counts rec's outcome and records it up to maxReportedResults; the counts -// stay authoritative when Results is truncated. Total is counted as records -// are read. +// stay authoritative when Results is truncated. Total is the run's record +// count. func (r *batchResult) add(rec *pendingRecord) { entry := recordResult{Index: rec.index} switch { case rec.reject != nil: r.Failed++ - entry.Error = rec.reject.Message + entry.Error, entry.ExceptionCode = rec.reject.Message, rec.reject.ExceptionCode case rec.duplicate: r.Duplicates++ entry.Duplicate = true @@ -473,14 +562,232 @@ func (r *batchResult) add(rec *pendingRecord) { } } -// writeAbort emits a whole-request failure response: the status and message, -// plus a Retry-After header when one is set (503 backpressure). Shared by the -// single-object and batch paths. +// insertShape turns the role's resolved insert grant into the projection of the +// table ClickHouse itself will enforce, plus the predicates the check clauses +// become. +// +// Columns is the allow/deny decision, answered by omitting the denied columns +// from the compiled schema: a record naming one is then refused per row with +// ClickHouse's own code 117 rather than by a Go walk over the record's keys. +// nil means the role may write every column, which compiles to no second +// handle at all. +// +// Defaults is the `_eq` auto-inject: the required value becomes the column's +// DEFAULT, so a record that omits it is filled and a record that supplies one +// still wins, and is then tested by the filter. An `_in` check has no single +// value to inject, so the column keeps the TABLE's own default and the filter +// tests that. +func (h *IngestHandler) insertShape( + ctx context.Context, + table, role string, + schema *discovery.TableSchema, + perms *policy.ResolvedPermissions, + guard *recordReject, +) (typelayer.RoleShape, []policy.Predicate, []string, *requestAbort) { + if perms == nil { + return typelayer.RoleShape{}, nil, nil, nil + } + + // Through the accessor, not a bare read: a bare read presents an empty map + // on an unresolved side and every check then passes vacuously. ok=false + // ABORTS the request; it must never be read as "no checks to run". + // + // Reachable for every table: nothing rejects an empty record before this + // point, because ClickHouse reads `{}` as every column taking its default. + checks, resolved := perms.CheckClauses() + if !resolved { + // ABORT, not a per-record reject: perms is resolved once per request, so + // this is true for every record or none. As a reject, a 10k-record batch + // would emit 10k ERROR lines and report 10k independent permission + // failures for one mis-wired grant. + slog.ErrorContext(ctx, "insert checks consulted on a grant resolved for another operation", + "table", table, "role", role) + return typelayer.RoleShape{}, nil, nil, &requestAbort{ + Status: http.StatusForbidden, + Message: "insert permissions were not resolved for this request", + } + } + + shape := typelayer.RoleShape{Columns: allowedInsertColumns(schema, perms)} + if guard != nil { + // The guard's own columns cannot be injected into or filtered on — that + // is what it refuses — so the shape stays bare and every record that + // parses gets the guard's rejection instead. + return shape, nil, nil, nil + } + + cols := slices.Sorted(maps.Keys(checks)) + preds := make([]policy.Predicate, 0, len(cols)) + for _, col := range cols { + switch v := checks[col].(type) { + case []any: + // An _in set. A nil/empty set is an unresolvable claim and matches + // nothing — Ingest fails every record's check without compiling + // anything (#224). + vals := make([]string, 0, len(v)) + for _, e := range v { + s, ok := scalarString(e) + if !ok { + vals = nil + break + } + vals = append(vals, s) + } + preds = append(preds, policy.Predicate{Column: col, Op: "in", Values: vals}) + default: + s, ok := scalarString(checks[col]) + if !ok { + // A required value with no string form cannot be expressed as a + // filter constant, and admitting the record would drop the check + // entirely. An empty Values list matches nothing. + preds = append(preds, policy.Predicate{Column: col, Op: "="}) + continue + } + if shape.Defaults == nil { + shape.Defaults = make(map[string]string, len(cols)) + } + shape.Defaults[col] = s + preds = append(preds, policy.Predicate{Column: col, Op: "=", Values: []string{s}}) + } + } + return shape, preds, cols, nil +} + +// allowedInsertColumns is the role's writable column set, or nil when it may +// write every column the table has. nil is the identity shape, which reuses the +// table's own compiled handle instead of a second one. +// +// Every column is asked through IsColumnAllowed so the allow/deny precedence +// stays in the one place that owns it; the computed kinds are included for the +// same reason, and typelayer keeps them whatever this list says (they are the +// server's to compute, and a MATERIALIZED expression over a dropped column would +// not compile at all). +func allowedInsertColumns(schema *discovery.TableSchema, perms *policy.ResolvedPermissions) []string { + allowed := make([]string, 0, len(schema.Columns)) + for _, c := range schema.Columns { + if perms.IsColumnAllowed(c.Name, true) { + allowed = append(allowed, c.Name) + } + } + if len(allowed) == len(schema.Columns) { + return nil + } + return allowed +} + +// scalarString renders a check clause's required value as the string the filter +// binds. Every filter parameter binds as {pN:String} whatever the column's +// declared type, so this is the only conversion the check path needs. +// +// The reflect.Kind test rather than a type switch is deliberate: policy marks a +// placeholder-free check value with its own string-kinded named type, which +// existed only for a Go-side numeric re-reading ClickHouse now answers. Naming +// the type here would keep it alive; asking for its kind works across its +// removal. +func scalarString(v any) (string, bool) { + if s, ok := v.(string); ok { + return s, true + } + rv := reflect.ValueOf(v) + if rv.Kind() == reflect.String { + return rv.String(), true + } + return "", false +} + +// roleTable resolves the compiled handle for this role's projection, or the 503 +// every record of this request gets instead. The type layer being down is an +// outage, never a verdict about the data: a caller must be able to retry the +// same body unchanged. +// +// An injected literal the column cannot read (`count UInt64 DEFAULT 'abc'`) is a +// compile refusal, ClickHouse code 6 — measured. That must not become a 503 for +// a policy that is simply unsatisfiable, so the shape is retried without its +// defaults: the check filter then judges the record as sent, which fails closed +// (an absent column takes the table default and the filter refuses it). The +// type layer logs the refusal once per generation and shape; this adds one +// rate-limited line saying what was done about it. +func (h *IngestHandler) roleTable(ctx context.Context, id tenant.ID, table string, shape typelayer.RoleShape) (*typelayer.Table, *requestAbort) { + if h.Types == nil { + h.logUnavailable(ctx, id, table, "no type layer is wired") + return nil, unavailableAbort() + } + tbl, err := h.Types.RoleTable(id, table, shape) + if err == nil { + return tbl, nil + } + if len(shape.Defaults) > 0 { + bare := typelayer.RoleShape{Columns: shape.Columns} + if t, bareErr := h.Types.RoleTable(id, table, bare); bareErr == nil { + if h.noticeDue("inject:" + id.String() + "/" + table) { + slog.WarnContext(ctx, "insert check value cannot be injected as a column default; records omitting it will fail the check", + "tenant", id, "table", table, "columns", slices.Sorted(maps.Keys(shape.Defaults)), "cause", err.Error()) + } + return t, nil + } + } + return nil, h.typesFailed(ctx, id, table, err) +} + +// typesFailed maps a type layer failure to the request's 503, logging its +// cause at most once a minute per tenant and table. +func (h *IngestHandler) typesFailed(ctx context.Context, id tenant.ID, table string, err error) *requestAbort { + cause := err.Error() + if un, ok := errors.AsType[*typelayer.Unavailable](err); ok { + cause = un.Cause + } + h.logUnavailable(ctx, id, table, cause) + return unavailableAbort() +} + +// unavailableAbort is the 503 for a type layer that cannot judge the tenant's +// table: a missing artifact for its ClickHouse line, a server zone this process +// cannot serve, a table that did not compile, or a tenant not bound yet. The +// body stays generic — the cause can name server paths and other tenants' +// zones, and is the operator's to read in the log. Retry-After is the schema +// hint's: the tenant's next schema refresh is what binds it again. +func unavailableAbort() *requestAbort { + return &requestAbort{ + Status: http.StatusServiceUnavailable, + Message: "ingest validation is unavailable", + RetryAfter: retryAfterSchema, + } +} + +// logUnavailable emits at most one line per tenant and table per minute. A +// missing artifact or a timezone mismatch persists until an operator acts, so +// the per-request line says nothing the first one didn't. +func (h *IngestHandler) logUnavailable(ctx context.Context, id tenant.ID, table, cause string) { + if h.noticeDue("unavailable:" + id.String() + "/" + table) { + slog.ErrorContext(ctx, "ingest type layer unavailable", "tenant", id, "table", table, "cause", cause) + } +} + +// noticeDue rate-limits a standing-condition notice to one line per key per +// minute, so a condition that persists until an operator acts does not bury the +// rest of the log under one line per request. +func (h *IngestHandler) noticeDue(key string) bool { + now := time.Now() + h.noticeMu.Lock() + defer h.noticeMu.Unlock() + if last, seen := h.noticeLast[key]; seen && now.Sub(last) < time.Minute { + return false + } + if h.noticeLast == nil { + h.noticeLast = make(map[string]time.Time) + } + h.noticeLast[key] = now + return true +} + +// writeAbort emits a whole-request failure response: the status, message and +// any error class and ClickHouse code, plus a Retry-After header when one is +// set. Shared by the single-object and batch paths. func writeAbort(w http.ResponseWriter, abort *requestAbort) { if abort.RetryAfter != "" { w.Header().Set("Retry-After", abort.RetryAfter) } - writeJSONError(w, abort.Status, abort.Message) + writeJSONErrorBody(w, abort.Status, errorBody{Error: abort.Message, Code: abort.Code, ExceptionCode: abort.ExceptionCode}) } // writeMaxBytesError writes a 413 if err is the inbound body-cap overflow and @@ -498,18 +805,22 @@ func writeMaxBytesError(w http.ResponseWriter, err error, limit int64) bool { // returns the rejection every record should get when they do not. // // A check the row cannot carry can never be enforced: the row holds one slot -// per INSERTABLE column, by position, so an auto-injected value for anything -// outside that set is dropped on the way out and the record inserts WITHOUT the -// value the policy requires — answering 200. Policy validation cannot catch -// this; it never sees the ClickHouse schema. Three ways in, all refused. +// per WIRE column, by position, so an injected value for anything outside that +// set is dropped on the way out and the record inserts WITHOUT the value the +// policy requires — answering 200. Policy validation cannot catch this; it never +// sees the ClickHouse schema. Three ways in, all refused. +// +// It is not redundant now that the compiled schema answers column policy: a +// Defaults entry for a column the shape cannot carry is a COMPILE refusal, so +// without this guard a mis-wired policy would be a 503 naming a ClickHouse +// internal rather than a 403 naming the column the operator has to fix. // // Evaluated here rather than per record because the condition is a property of -// (table, role, policy) and is identical for every record in the request — the -// same reasoning as the !resolved abort in prepareRecord. Doing it per record -// would emit one ERROR line per record for a single mis-wired policy, which on -// a 16 MiB body of small records is ~1.2M lines. The reject is still returned -// per record, so a batch reports each record's own cause: one that SUPPLIES the -// column fails schema validation first, with a different message. +// (table, role, policy) and is identical for every record in the request. Doing +// it per record would emit one ERROR line per record for a single mis-wired +// policy, which on a 16 MiB body of small records is ~1.2M lines. The reject is +// still returned per record, so a batch reports each record's own cause: one +// that SUPPLIES the column is refused by ClickHouse first, with its own code. func (h *IngestHandler) policyCheckGuard( ctx context.Context, table, role string, @@ -518,7 +829,7 @@ func (h *IngestHandler) policyCheckGuard( ) *recordReject { checks, resolved := perms.CheckClauses() if !resolved { - return nil // the !resolved abort in prepareRecord owns this case + return nil // the !resolved abort in insertShape owns this case } // Sorted, and every offender — not the first one a map range happens to @@ -566,7 +877,7 @@ func (h *IngestHandler) policyCheckGuard( } } -// pendingRecord is one record between prepareRecord and its outcome. +// pendingRecord is one record between prepareVerdict and its outcome. type pendingRecord struct { index int // 1-based position in a batch reject *recordReject // non-nil: the record is bad and is not published @@ -580,128 +891,37 @@ type pendingRecord struct { duplicate bool } -// prepareRecord runs the per-record half of the pipeline shared by the -// single-object and batch ingest paths: schema validation → column/check -// permission enforcement (with claim-derived auto-injection) → timestamp -// canonicalization → dedupe id resolution → encoding. Reserving, publishing -// and committing happen per window, in ingestWindow. The table-level insert -// grant is checked once by the caller before any record is processed, so perms -// here drives only the per-column and per-row checks (it is nil when no policy -// store is configured). data may be mutated to auto-inject check-clause values. +// prepareVerdict turns the i-th record's verdict into its pending outcome: +// ClickHouse's refusal or decline, the check guard, the check answer, then the +// dedupe id and the encoded envelope of the row ClickHouse's own writer +// produced. Reserving, publishing and committing happen per window, in +// ingestWindow. +// +// The order is the one the type layer imposes, and is documented: a record +// that both fails to parse and violates a check clause reports the PARSE +// error — chtypes answers the check only for a record it accepted. Nothing is +// published either way, so no enforcement is lost. +// +// Dedupe runs AFTER validation deliberately — the id must be claimed only for +// what is published, or a record ClickHouse refuses would burn its id and a +// corrected retry would be swallowed as a duplicate. // // A record the rest of a batch may proceed past comes back with reject set; // abort non-nil is a whole-request failure the caller stops and returns. -func (h *IngestHandler) prepareRecord( - ctx context.Context, - store *settings.Store, - table, scope string, - schema *discovery.TableSchema, - perms *policy.ResolvedPermissions, - role string, - data map[string]any, - now time.Time, - checkGuard *recordReject, -) (rec pendingRecord, abort *requestAbort) { - if err := h.validator().Validate(schema, data); err != nil { - slog.WarnContext(ctx, "schema validation failed", "error", err, "table", table) - return pendingRecord{reject: &recordReject{Status: http.StatusBadRequest, Message: err.Error()}}, nil - } - - // DEEP AUTH: column-level allow/deny + check clauses. - if perms != nil { - for col := range data { - if !perms.IsColumnAllowed(col, true) { - slog.WarnContext(ctx, "column insertion forbidden", "column", col, "role", role) - return pendingRecord{reject: &recordReject{ - Status: http.StatusForbidden, - Message: fmt.Sprintf("column %q not allowed for insert", col), - }}, nil - } - } - // Through the accessor, not a bare read. The check loop iterates a side's - // map rather than asking about a column, so IsColumnAllowed cannot cover it, - // and a bare read presents an empty map on an unresolved side — every check - // then passes vacuously. ok=false ABORTS the request; it must never be read - // as "no checks to run". - // - // Reachable, unlike the query path's bare reads: discovery.Validate only - // requires a column that is neither nullable nor defaulted, so a table whose - // columns are all nullable or all defaulted accepts `{}`, and the column loop - // above then runs zero times. - checks, resolved := perms.CheckClauses() - if !resolved { - // ABORT, not a per-record reject: perms is resolved once per request, - // so this is true for every record or none. As a reject, a 10k-record - // batch would emit 10k ERROR lines and report 10k independent - // permission failures for one mis-wired grant. - slog.ErrorContext(ctx, "insert checks consulted on a grant resolved for another operation", - "table", table, "role", role) - return pendingRecord{}, &requestAbort{ - Status: http.StatusForbidden, - Message: "insert permissions were not resolved for this request", - } - } - for col, requiredVal := range checks { - // The guard for this is evaluated ONCE per request in - // policyCheckGuard (see Handle) and only consulted here: the condition - // is a property of (table, role, policy), identical for every record, - // so evaluating it per record would emit one ERROR line per record for - // a single mis-wired policy — the same amplification the !resolved - // abort above exists to avoid. The REJECT is still per record, because - // a record that supplies the column fails schema validation first with - // a different message, and a batch should report each its own cause. - if checkGuard != nil { - return pendingRecord{reject: checkGuard}, nil - } - // A []any value is an _in check: the inserted value must be present and - // one of the allowed set. Unlike the scalar _eq case there is no single - // value to auto-inject, so an absent column fails closed. - if set, isSet := requiredVal.([]any); isSet { - actual, ok := data[col] - if !ok || !h.checker().InSet(actual, set) { - slog.WarnContext(ctx, "check clause failed", "column", col, "allowed", set, "actual", actual, "present", ok) - return pendingRecord{reject: &recordReject{ - Status: http.StatusForbidden, - Message: fmt.Sprintf("check failed for column %q", col), - }}, nil - } - continue - } - if actual, ok := data[col]; ok { - // Both sides canonicalize through policy.CanonicalScalar, so a - // numeric insert value matches a numeric claim by value, not by - // spelling (payload 1.0 vs claim 1), and a value with no canonical - // form (object/array/null) matches nothing. A policy.LiteralValue — - // a placeholder-free check value, which carries no JSON type — - // additionally matches by its numeric reading, so `_eq: "1.0"` - // accepts an inserted 1.0 as well as an inserted "1.0". The type is - // what scopes that second reading to author-written literals: a - // claim-derived value arrives as a plain string and never gains a - // reading the token's own JSON type didn't give it. - if !h.checker().Matches(actual, requiredVal) { - slog.WarnContext(ctx, "check clause failed", "column", col, "expected", requiredVal, "actual", actual) - return pendingRecord{reject: &recordReject{ - Status: http.StatusForbidden, - Message: fmt.Sprintf("check failed for column %q", col), - }}, nil - } - } else { - // Auto-inject the required value if not provided — as a plain - // string: a LiteralValue must not leak its named type into the - // published payload, where downstream type switches (timestamp - // canonicalization's `case string`) would silently miss it. - if lit, isLit := requiredVal.(policy.LiteralValue); isLit { - data[col] = string(lit) - } else { - data[col] = requiredVal - } - } - } +func (h *IngestHandler) prepareVerdict(ctx context.Context, run *ingestRun, i int) (rec pendingRecord, abort *requestAbort) { + verdict := verdictAt(run.batch, i) + if !verdict.Accepted { + logVerdict(ctx, run.table, verdict) + return pendingRecord{reject: verdictReject(verdict)}, nil + } + if run.checkGuard != nil { + return pendingRecord{reject: run.checkGuard}, nil + } + if verdict.CheckReason != "" { + slog.WarnContext(ctx, "check clause failed", "columns", run.checkColumns, + "reason", verdict.CheckReason, "cause", verdict.Message, "table", run.table) + return pendingRecord{reject: checkReject(verdict.CheckReason, run.checkColumns)}, nil } - - // Canonicalize timestamps to RFC 3339 UTC (#372; fail-open — #381's row-filter - // enforces) after the permission checks: check clauses keep pre-#372 semantics. - h.validator().CanonicalizeTimestamps(schema, data) // Optional deduplication. The dedupe settings resolve per record // from one snapshot (table override → global; the settings directory @@ -711,47 +931,42 @@ func (h *IngestHandler) prepareRecord( // in ingestWindow, once every record of the window is encoded, so nothing // but the publish can fail while the claim is held. if h.Dedup != nil && h.DedupeSettings != nil { - if dd := h.DedupeSettings(store, table); dd.Enabled { + if dd := h.DedupeSettings(run.store, run.table); dd.Enabled { idField := dd.IDField - // An explicit null is as missing as an absent key (#370): fmt.Sprint - // would make every null "", one id for every such record. - if idVal, ok := data[idField]; ok && idVal != nil { - rec.key = &dedupe.Key{Table: table, ID: fmt.Sprint(idVal)} + if id, ok := eventIDAt(verdict.Line, slices.Index(run.wire, idField)); ok { + rec.key = &dedupe.Key{Table: run.table, ID: id} rec.retention = dd.Retention } else { - dedupeMissingIDCounter.Add(ctx, 1, metric.WithAttributes(attribute.String("table", table))) + dedupeMissingIDCounter.Add(ctx, 1, metric.WithAttributes(attribute.String("table", run.table))) if dd.RequireID { - slog.WarnContext(ctx, "dedupe id_field missing or null; rejecting", "id_field", idField, "table", table) + slog.WarnContext(ctx, "dedupe id_field missing or null; rejecting", "id_field", idField, "table", run.table) return pendingRecord{reject: &recordReject{ Status: http.StatusBadRequest, Message: fmt.Sprintf("missing dedupe id field %q", idField), }}, nil } - slog.WarnContext(ctx, "dedupe id_field missing or null; publishing without idempotency", "id_field", idField, "table", table) + slog.WarnContext(ctx, "dedupe id_field missing or null; publishing without idempotency", "id_field", idField, "table", run.table) } } } - // Render the record positionally against the table's declaration order. The - // column names ride alongside in the envelope rather than in the row, so a - // batch of rows for one table carries the names once — and the reader can - // tell a schema change mid-stream from a reordering. - cols := schema.InsertableColumns() - row, err := ingest.EncodeCompactRow(cols, data) - if err != nil { - slog.ErrorContext(ctx, "failed to encode compact row", "error", err, "table", table) - return pendingRecord{}, &requestAbort{Status: http.StatusInternalServerError, Message: "marshal failed"} - } - + // The row travels POSITIONALLY as the bytes ClickHouse's own writer + // produced, with the column names alongside rather than in the row: a batch + // of rows for one table carries the names once, and the reader can tell a + // schema change mid-stream from a reordering. They are the ROLE's wire + // columns, so a role that may not write every column publishes a shorter + // row with a matching name list — the worker already groups a flush by + // column signature, so that is one more group, not a new code path. evt := ingest.EventMessage{ - TableName: table, - Scope: scope, - ReceivedTimestamp: now.Format(time.RFC3339Nano), + TableName: run.table, + Scope: run.scope, + ReceivedTimestamp: run.now.Format(time.RFC3339Nano), Format: ingest.FormatJSONCompactEachRow, - Columns: schema.InsertableColumnNames(), - Row: row, + Columns: run.wire, + Row: json.RawMessage(verdict.Line), } + var err error rec.payload, err = json.Marshal(evt) if err != nil { slog.ErrorContext(ctx, "failed to marshal event message", "error", err) @@ -760,6 +975,72 @@ func (h *IngestHandler) prepareRecord( return rec, nil } +// verdictAt reads the verdict for the i-th record of a batch. A verdict the +// type layer did not return is a decline, never an acceptance: a missing answer +// must not publish a row nobody ruled on. +func verdictAt(batch typelayer.Batch, i int) typelayer.RowVerdict { + if i < len(batch.Rows) { + return batch.Rows[i] + } + return typelayer.RowVerdict{Declined: true, Message: "no verdict was returned for this record"} +} + +// verdictReject maps a verdict that is not an acceptance to the record's +// rejection. A refusal is ClickHouse's own answer about the data — 400, with +// its code. A decline is the validation engine failing to answer at all, which +// is never the caller's fault and must never be dressed as a 400: 422 says "we +// could not judge this", so a retry is meaningful and a client cannot learn +// from it that its payload was wrong. +func verdictReject(v typelayer.RowVerdict) *recordReject { + if v.Declined { + return &recordReject{ + Status: http.StatusUnprocessableEntity, + Message: "validation engine declined: " + v.Message, + } + } + return &recordReject{Status: http.StatusBadRequest, Message: v.Message, ExceptionCode: v.Code} +} + +// checkReject maps one check verdict that is not a definite true. "The data says +// no" is a 403; "we could not tell" is a 422, fail closed either way. +// +// The filter is AND-joined over every check clause, so a false verdict does not +// name which clause failed — with a single clause the column is unambiguous, and +// with several the message names the set that was tested rather than inventing +// an attribution. +func checkReject(reason string, cols []string) *recordReject { + if reason == typelayer.ReasonFilter { + return &recordReject{Status: http.StatusForbidden, Message: "check failed for " + columnList(cols)} + } + return &recordReject{ + Status: http.StatusUnprocessableEntity, + Message: "validation engine declined: the insert check for " + columnList(cols) + " could not be evaluated", + } +} + +// columnList renders a check's column set for a rejection message. +func columnList(cols []string) string { + quoted := make([]string, len(cols)) + for i, c := range cols { + quoted[i] = fmt.Sprintf("%q", c) + } + if len(cols) == 1 { + return "column " + quoted[0] + } + return "columns " + strings.Join(quoted, ", ") +} + +// logVerdict records a record ClickHouse would not take. A refusal carries its +// code so an operator can look it up without parsing the message; a decline is +// an ERROR because it is the engine, not the data, that failed. +func logVerdict(ctx context.Context, table string, v typelayer.RowVerdict) { + if v.Declined { + slog.ErrorContext(ctx, "validation engine declined a record", "reason", v.Message, "table", table) + return + } + slog.WarnContext(ctx, "schema validation failed", "error", v.Message, "exception_code", v.Code, "table", table) +} + // ingestWindow reserves, publishes and commits one window of prepared // records, in three phases: one Reserve for every keyed record, the publishes // in record order, one Commit for every claim published. Rejected and @@ -968,46 +1249,3 @@ func releaseClaims(ctx context.Context, dd dedupe.Deduplicator, claims []dedupe. slog.WarnContext(ctx, "dedupe release failed; the ids lapse with their lease", "error", err) } } - -// checkValueMatches decides insert-check equality: the payload value must -// have a canonical scalar form (object/array/null match nothing) equal to the -// required value's canonical form. A policy.LiteralValue — and only that type, -// which Evaluate reserves for placeholder-free check values — also matches by -// its canonical numeric reading (policy.CanonicalNumericLiteral), so a static -// `_eq: "1.0"` accepts an inserted 1.0 and an inserted "1.0" alike. -func checkValueMatches(actual, required any) bool { - actualStr, hasForm := policy.CanonicalScalar(actual) - if !hasForm { - return false - } - if lit, isLit := required.(policy.LiteralValue); isLit { - if actualStr == string(lit) { - return true - } - n, ok := policy.CanonicalNumericLiteral(string(lit)) - return ok && actualStr == n - } - requiredStr, ok := policy.CanonicalScalar(required) - return ok && actualStr == requiredStr -} - -// valueInSet reports whether v matches any member of set, comparing by -// canonical string form (policy.CanonicalScalar) to mirror the scalar check's -// claim-derived equality — a JSON number in the insert body matches a -// claim-derived value by value, not spelling, and a v with no canonical form -// (object/array/null) is a member of no set. _in members never take the -// LiteralValue numeric reading: an _in set is claim-derived by design, and a -// placeholder-free _in template is a degenerate one-element set that keeps -// spelling equality. -func valueInSet(v any, set []any) bool { - vs, ok := policy.CanonicalScalar(v) - if !ok { - return false - } - for _, s := range set { - if ss, ok := policy.CanonicalScalar(s); ok && ss == vs { - return true - } - } - return false -} diff --git a/internal/api/ingest_formats_test.go b/internal/api/ingest_formats_test.go new file mode 100644 index 00000000..9fcac5e6 --- /dev/null +++ b/internal/api/ingest_formats_test.go @@ -0,0 +1,411 @@ +package api + +import ( + "fmt" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/auth" + "github.com/Wave-RF/WaveHouse/internal/policy" + "github.com/Wave-RF/WaveHouse/internal/testutil" +) + +// TestIngest_JSONArray_CompactWithOneBadRecord is the regression guard for the +// whole reason the depth-1 comma rewrite exists. +// +// Measured: a SINGLE-LINE array with one bad record makes chtypes answer +// Outcome=rejected with no exported bytes — the records that parsed perfectly +// are lost with it. That breaks #195's promise that one bad record never +// obscures the rest of the batch. Newline-framing the elements restores it. +func TestIngest_JSONArray_CompactWithOneBadRecord(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "application/json", + `[{"page":"/a"},{"page":"/b","nope":1},{"page":"/c"}]`))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 3, resp.Total) + assert.Equal(t, 2, resp.Succeeded) + assert.Equal(t, 1, resp.Failed) + require.Len(t, resp.Results, 3) + assert.True(t, resultAt(t, resp, 1).Ok) + assert.Equal(t, 117, resultAt(t, resp, 2).ExceptionCode) + assert.True(t, resultAt(t, resp, 3).Ok) + require.Len(t, pub.Messages, 2, "the siblings of a refused record still publish") + assert.Equal(t, "/a", publishedRow(t, pub.Messages[0].Data)["page"]) + assert.Equal(t, "/c", publishedRow(t, pub.Messages[1].Data)["page"]) +} + +// TestIngest_CSV: CSV is header-less and POSITIONAL in the table's declaration +// order — the wire columns, which is declaration order minus MATERIALIZED, ALIAS +// and EPHEMERAL. Every wire column must be present; an empty field takes the +// column's DEFAULT. +func TestIngest_CSV(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + // clicks is page, button, count, event_id, org_id. + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv", + "\"/a\",\"buy\",3,\"e1\",\"acme\"\n\"/b\",\"sell\",4,\"e2\",\"acme\"\n"))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 2, resp.Total) + assert.Equal(t, 2, resp.Succeeded) + require.Len(t, pub.Messages, 2) + row := publishedRow(t, pub.Messages[0].Data) + assert.Equal(t, "/a", row["page"]) + assert.Equal(t, "buy", row["button"]) + assert.Equal(t, float64(3), row["count"]) + assert.Equal(t, "acme", row["org_id"]) +} + +// TestIngest_CSV_PositionalContract pins the three ways a producer gets the +// positional contract wrong, each with ClickHouse's own code, under +// `header=absent` (strictly positional, detection off). A header line there is +// not a header: it is one record that fails to parse, and the data rows around +// it still ingest. +func TestIngest_CSV_PositionalContract(t *testing.T) { + t.Parallel() + for _, tt := range []struct { + name string + body string + ok int + bad int + }{ + {"a short row is a per-record failure", "\"/a\",\"buy\"\n\"/b\",\"sell\",4,\"e2\",\"acme\"\n", 1, 1}, + {"an extra field is a per-record failure", "\"/a\",\"buy\",3,\"e1\",\"acme\",99\n", 0, 1}, + {"a header line is one failed record, not a header", "page,button,count,event_id,org_id\n\"/b\",\"sell\",4,\"e2\",\"acme\"\n", 1, 1}, + {"an empty field takes the column default", "\"/a\",\"buy\",3,\"e1\",\n", 1, 0}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv; header=absent", tt.body))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, tt.ok, resp.Succeeded) + assert.Equal(t, tt.bad, resp.Failed) + assert.Len(t, pub.Messages, tt.ok) + for _, r := range resp.Results { + if r.Error != "" { + assert.NotZero(t, r.ExceptionCode, "a parser refusal carries ClickHouse's code") + } + } + }) + } +} + +// TestIngest_BareCSV_AutoDetectsHeader: a `text/csv` with no header parameter +// is ClickHouse's default CSV, which consumes a first line that spells the +// column names as a header. The indices and total count only data rows; the +// same body under header=absent rejects that line at index 1 (code 27); a bare +// body with no header line ingests every row. +func TestIngest_BareCSV_AutoDetectsHeader(t *testing.T) { + t.Parallel() + const header = "page,button,count,event_id,org_id\n" + const rows = "\"/a\",\"buy\",3,\"e1\",\"acme\"\n\"/b\",\"sell\",4,\"e2\",\"acme\"\n" + + t.Run("bare consumes the header line", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv", header+rows))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 2, resp.Total, "the detected header is not a record") + assert.Equal(t, 2, resp.Succeeded) + require.Len(t, pub.Messages, 2) + assert.Equal(t, "/a", publishedRow(t, pub.Messages[0].Data)["page"]) + }) + + t.Run("header=absent rejects the header line", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv; header=absent", header+rows))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 3, resp.Total) + assert.Equal(t, 2, resp.Succeeded) + assert.Equal(t, 27, resultAt(t, resp, 1).ExceptionCode, "the first line is a record ClickHouse's positional reader refuses") + assert.Len(t, pub.Messages, 2) + }) + + t.Run("bare with no header line ingests every row", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv", rows))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 2, resp.Total) + assert.Equal(t, 2, resp.Succeeded) + assert.Len(t, pub.Messages, 2) + }) +} + +// TestIngest_TSV is CSV's tab-separated twin, with the same positional contract +// and ClickHouse's own \N for null. +func TestIngest_TSV(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/tab-separated-values", + "/a\tbuy\t3\te1\tacme\n/b\tsell\tnot-a-number\te2\tacme\n"))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 2, resp.Total) + assert.Equal(t, 1, resp.Succeeded) + assert.Equal(t, 1, resp.Failed) + assert.Equal(t, 27, resultAt(t, resp, 2).ExceptionCode) + require.Len(t, pub.Messages, 1) + assert.Equal(t, "/a", publishedRow(t, pub.Messages[0].Data)["page"]) +} + +// TestIngest_CSV_IsAlwaysABatch: the positional formats have no arity question — +// a one-row CSV still answers with the per-record envelope, never {"ok":true}. +func TestIngest_CSV_IsAlwaysABatch(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv", "\"/a\",\"buy\",3,\"e1\",\"acme\"\n"))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 1, resp.Total) + assert.Equal(t, 1, resp.Succeeded) +} + +// TestIngest_CSV_EmptyBody names the format in the 400, like the NDJSON one. +func TestIngest_CSV_EmptyBody(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv", " \n "))) + + assert.Equal(t, http.StatusBadRequest, w.Code) + assert.Contains(t, jsonErrorMessage(t, w), "empty csv body") + assert.Empty(t, pub.Messages) +} + +// TestIngest_CSVWithNames: `text/csv; header=present` reads the first line as +// the column names, in any order. The header is not a record, so indices +// count data lines; a column the header omits takes its DEFAULT. +func TestIngest_CSVWithNames(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv; header=present", + "org_id,page,count\nacme,/a,3\nacme,/b,not-a-number\nbeta,/c,5\n"))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 3, resp.Total, "the header line is not a record") + assert.Equal(t, 2, resp.Succeeded) + assert.True(t, resultAt(t, resp, 1).Ok) + assert.False(t, resultAt(t, resp, 2).Ok) + assert.NotZero(t, resultAt(t, resp, 2).ExceptionCode, "a parser refusal carries ClickHouse's code") + assert.True(t, resultAt(t, resp, 3).Ok) + require.Len(t, pub.Messages, 2) + row := publishedRow(t, pub.Messages[0].Data) + assert.Equal(t, "/a", row["page"]) + assert.Equal(t, float64(3), row["count"]) + assert.Equal(t, "acme", row["org_id"]) + assert.Equal(t, "", row["button"], "a column the header omits takes its DEFAULT") + assert.Equal(t, "/c", publishedRow(t, pub.Messages[1].Data)["page"]) +} + +// TestIngest_TSVWithNames is the tab-separated twin. +func TestIngest_TSVWithNames(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/tab-separated-values; header=present", + "count\tpage\n7\t/a\n"))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 1, resp.Total) + assert.Equal(t, 1, resp.Succeeded) + require.Len(t, pub.Messages, 1) + assert.Equal(t, float64(7), publishedRow(t, pub.Messages[0].Data)["count"]) +} + +// TestIngest_WithNames_HeaderRefusals: a header ClickHouse refuses is a +// verdict on the body, not on a record — a whole-request 400 with its own +// code, and nothing published. A header alone is zero records. +func TestIngest_WithNames_HeaderRefusals(t *testing.T) { + t.Parallel() + for name, tc := range map[string]struct { + ct, body, mention string + }{ + "unknown column": {"text/csv; header=present", "page,extra\n/a,1\n", "extra"}, + "repeated column": {"text/csv; header=present", "page,page\n/a,/b\n", "page"}, + "tsv unknown": {"text/tab-separated-values; header=present", "page\textra\n/a\t1\n", "extra"}, + } { + t.Run(name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", tc.ct, tc.body))) + + require.Equal(t, http.StatusBadRequest, w.Code, "body=%s", w.Body.String()) + msg, code := errorAndCode(t, w) + assert.Equal(t, 117, code) + assert.Equal(t, codeCHRejected, errorClass(t, w), "a whole-request parser refusal carries the class too") + assert.Contains(t, msg, tc.mention) + assert.Empty(t, pub.Messages) + }) + } + + t.Run("header only", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv; header=present", "page,count\n"))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, 0, decodeBatchResult(t, w).Total) + assert.Empty(t, pub.Messages) + }) + + t.Run("empty body names the format", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "text/csv; header=present", " \n"))) + + assert.Equal(t, http.StatusBadRequest, w.Code) + assert.Contains(t, jsonErrorMessage(t, w), "empty csvwithnames body") + }) +} + +// TestIngest_WithNames_RoleProjection: the header is read against the ROLE's +// compiled schema, so a denied column in it is ClickHouse's 117 for the whole +// body, an `_eq` check column the header omits is filled by its injected +// DEFAULT, and a record that supplies another value fails the check (403). +func TestIngest_WithNames_RoleProjection(t *testing.T) { + t.Parallel() + required := "acme" + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + h.PolicySource = staticPolicy(&policy.Policy{Tables: map[string]policy.TablePolicy{ + "clicks": {"writer": {Insert: &policy.InsertPermissions{ + DenyColumns: []string{"count"}, + Check: map[string]policy.Filter{"org_id": {Eq: &required}}, + }}}, + }}) + send := func(body string) *httptest.ResponseRecorder { + req := rawIngestRequest(t, "clicks", "text/csv; header=present", body) + req = req.WithContext(auth.WithRole(req.Context(), "writer")) + w := httptest.NewRecorder() + h.Handle(w, withTenant(req)) + return w + } + + w := send("page,count\n/a,1\n") + require.Equal(t, http.StatusBadRequest, w.Code, "body=%s", w.Body.String()) + _, code := errorAndCode(t, w) + assert.Equal(t, 117, code) + assert.Empty(t, pub.Messages) + + w = send("page,org_id\n/a,acme\n/b,evil\n") + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.True(t, resultAt(t, resp, 1).Ok) + assert.Contains(t, resultAt(t, resp, 2).Error, `check failed for column "org_id"`) + + w = send("page\n/c\n") + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.True(t, resultAt(t, decodeBatchResult(t, w), 1).Ok) + require.Len(t, pub.Messages, 2) + assert.Equal(t, "acme", publishedRow(t, pub.Messages[1].Data)["org_id"], "the omitted check column took its injected DEFAULT") +} + +// TestIngest_LargeBatch_IndicesStayContiguous guards the one thing removing the +// handler's 500-record chunking could plausibly break. Verdicts used to be +// gathered per chunk and stitched back together by a staging slice; they are +// now one array from one Ingest call, indexed directly. A batch that straddles +// the old boundary must still report 1-based indices in order, with each +// record's own outcome — an off-by-one there would attribute a refusal to the +// wrong record, which the response gives a caller no way to detect. +func TestIngest_LargeBatch_IndicesStayContiguous(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + const n = 600 + bad := map[int]bool{1: true, 250: true, 500: true, 501: true, n: true} // 1-based + var b strings.Builder + b.WriteByte('[') + for i := 1; i <= n; i++ { + if i > 1 { + b.WriteByte(',') + } + if bad[i] { + fmt.Fprintf(&b, `{"page":"/p%d","nope":%d}`, i, i) + continue + } + fmt.Fprintf(&b, `{"page":"/p%d"}`, i) + } + b.WriteByte(']') + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "application/json", b.String()))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + require.Equal(t, n, resp.Total) + assert.Equal(t, len(bad), resp.Failed) + assert.Equal(t, n-len(bad), resp.Succeeded) + require.Len(t, resp.Results, n) + for i, r := range resp.Results { + require.Equal(t, i+1, r.Index, "results must be 1-based and in order") + if bad[i+1] { + assert.Equal(t, 117, r.ExceptionCode, "record %d", i+1) + continue + } + assert.True(t, r.Ok, "record %d", i+1) + } + require.Len(t, pub.Messages, n-len(bad)) + // The published rows are the accepted records, in order, with the refused + // ones simply absent — so record 502 sits four slots earlier than its index + // (records 1, 250, 500 and 501 were refused before it). Spot-checking a row + // on the far side of the old chunk boundary is what would catch a verdict + // misattributed across it. + assert.Equal(t, "/p502", publishedRow(t, pub.Messages[502-1-4].Data)["page"]) + assert.Equal(t, "/p2", publishedRow(t, pub.Messages[0].Data)["page"], "record 1 was refused") + assert.Equal(t, "/p599", publishedRow(t, pub.Messages[len(pub.Messages)-1].Data)["page"], "record 600 was refused") +} diff --git a/internal/api/ingest_framing.go b/internal/api/ingest_framing.go new file mode 100644 index 00000000..d4ebc67e --- /dev/null +++ b/internal/api/ingest_framing.go @@ -0,0 +1,191 @@ +package api + +import "encoding/json" + +// Framing: everything ingest reads out of the bytes it handles, the request +// body and the rows ClickHouse exported from it. It is deliberately small — +// single-pass scanners, no decoder — because the whole point of the type layer +// is that ClickHouse's parser reads the records and Go does not. + +// reframeArray turns a top-level JSON array into the newline-framed body +// chtypes reads per record, IN PLACE, and reports how many elements it holds. +// +// It exists because of a measured cliff: a SINGLE-LINE array with one bad +// record loses the whole batch — chtypes answers Outcome=rejected with no +// exported bytes, so the records that parsed perfectly are lost too (pinned by +// TestIngest_JSONArray_CompactWithOneBadRecord). The same +// records newline-separated skip the bad one and export the rest. Rewriting the +// depth-1 commas to newlines restores per-record salvage (#195's promise) for +// 0.7 ms per 617 KB, with no decode and no copy. +// +// The scan is string- and escape-aware, so a comma or a bracket inside a value +// is untouched, and it runs ONLY when the declared format is the JSON family and +// the first non-whitespace byte is '['. It must not run on anything else: a bare +// object's own commas are at depth 1 and rewriting them destroys the record +// (measured). +// +// Three substitutions, all in place and all the same length: +// +// - a depth-1 comma becomes a newline — the framing itself; +// - the OUTER '[' and ']' become spaces. Commas alone are not enough: +// measured, a bad LAST record still loses the whole batch, because the +// closing bracket shares that record's line and the reader cannot resync +// past it. JSONEachRow needs no brackets, so removing them costs nothing and +// makes every position salvageable, first and last included; +// - every other newline outside a string becomes a space. Not cosmetic: +// typelayer.Ingest pads its verdict list out to the body's newline count, so +// a pretty-printed array would come back with one phantom "no verdict" +// record per line of layout. This leaves exactly elements-1 newlines. +// +// A raw newline inside a string is illegal JSON, so leaving those alone costs +// nothing and keeps the caller's bytes the caller's. +// +// ok is false when the brackets do not balance — a truncated upload, or a +// structural syntax error — which is a whole-request 400. Nothing is published +// from a body we cannot frame. +func reframeArray(b []byte) (elements int, ok bool) { + depth, commas := 0, 0 + sawValue := false + inStr, esc := false, false + for i := range b { + c := b[i] + switch { + case esc: + esc = false + case inStr && c == '\\': + esc = true + case c == '"': + inStr = !inStr + sawValue = sawValue || depth >= 1 + case inStr: + case c == '[' || c == '{': + sawValue = sawValue || depth >= 1 + depth++ + if c == '[' && depth == 1 { + b[i] = ' ' + } + case c == ']' || c == '}': + depth-- + if c == ']' && depth == 0 { + b[i] = ' ' + } + case c == ',' && depth == 1: + b[i] = '\n' + commas++ + case c == '\n' || c == '\r': + b[i] = ' ' + case c == ' ' || c == '\t': + default: + sawValue = sawValue || depth >= 1 + } + } + if depth != 0 || inStr { + return 0, false + } + if !sawValue { + return 0, true // `[]`, possibly with whitespace inside + } + return commas + 1, true +} + +// cellAt returns the k-th top-level cell of one JSONCompactEachRow line — a +// `[v0, v1, …]` array as ClickHouse's own writer produced it — without decoding +// the row. Leading and trailing whitespace around the cell is trimmed; the cell +// itself is returned verbatim, still JSON-encoded. +// +// This is how the dedupe id is read (eventIDAt): the id column's position in +// the table's wire columns is known, so the value is a byte span rather than a +// map lookup. The scanner is the same string- and escape-aware shape as +// reframeArray, so a comma or a bracket inside a value cannot end a cell. +func cellAt(line []byte, k int) ([]byte, bool) { + if k < 0 { + return nil, false + } + depth, idx, start := 0, 0, -1 + inStr, esc := false, false + for i := range line { + c := line[i] + switch { + case esc: + esc = false + continue + case inStr && c == '\\': + esc = true + continue + case c == '"': + inStr = !inStr + case inStr: + case c == '[' || c == '{': + depth++ + if depth == 1 { + start = i + 1 + continue + } + case c == ']' || c == '}': + depth-- + if depth == 0 { + if idx == k && start >= 0 { + return trimSpaceBytes(line[start:i]), true + } + return nil, false + } + case c == ',' && depth == 1: + if idx == k { + return trimSpaceBytes(line[start:i]), true + } + idx++ + start = i + 1 + continue + } + } + return nil, false +} + +// trimSpaceBytes drops ASCII layout around a cell. bytes.TrimSpace would also +// do it, but this stays byte-exact about which bytes count as layout in a +// JSONCompactEachRow line (the writer emits ", " between cells) and allocates +// nothing. +func trimSpaceBytes(b []byte) []byte { + i, j := 0, len(b) + for i < j && (b[i] == ' ' || b[i] == '\t' || b[i] == '\n' || b[i] == '\r') { + i++ + } + for j > i && (b[j-1] == ' ' || b[j-1] == '\t' || b[j-1] == '\n' || b[j-1] == '\r') { + j-- + } + return b[i:j] +} + +// eventIDAt reads the dedupe id out of an exported row by POSITION — the id +// column's index in the wire columns the row was exported with — so no record +// is decoded for it. idx is -1 when the column is not on the wire at all. +// +// Consequences worth knowing, all documented: +// - the key is the STORED value, not the caller's spelling: `256` into a +// UInt8 keys on `0`, and a DateTime keys on ClickHouse's rendering. For the +// documented case — a string id — the two are identical. +// - "missing" means "the row carries no value". A `null` cell is missing, as +// an explicit null always has been (#370): keying on its spelling would make +// every null one id. So is an empty string, which is what an omitted +// `event_id String` stores. A numeric id column cannot distinguish an +// omitted 0 from a supplied one. +func eventIDAt(line []byte, idx int) (string, bool) { + if idx < 0 { + return "", false + } + cell, ok := cellAt(line, idx) + if !ok || len(cell) == 0 || string(cell) == "null" { + return "", false + } + if cell[0] == '"' { + // One scalar string, not the record: the cell is JSON-encoded by + // ClickHouse's own writer (it escapes "/" as "\/"), so Go's own + // string-literal unquoting would refuse it. + var s string + if err := json.Unmarshal(cell, &s); err != nil || s == "" { + return "", false + } + return s, true + } + return string(cell), true +} diff --git a/internal/api/ingest_framing_test.go b/internal/api/ingest_framing_test.go new file mode 100644 index 00000000..c400d1bd --- /dev/null +++ b/internal/api/ingest_framing_test.go @@ -0,0 +1,173 @@ +package api + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// TestReframeArray covers the one place ingest still looks at the body's own +// bytes. The rewrite is string- and escape-aware and runs ONLY on a declared +// JSON body whose first non-whitespace byte is '[' — the gate matters as much as +// the scan, because a bare object's commas are at depth 1 too and rewriting them +// destroys the record (TestReframeArray_DestroysWhatItMustNotSee). +func TestReframeArray(t *testing.T) { + t.Parallel() + for _, tt := range []struct { + name string + body string + want string + count int + ok bool + }{ + { + name: "compact array becomes one record per line", + body: `[{"a":1},{"a":2},{"a":3}]`, + want: " {\"a\":1}\n{\"a\":2}\n{\"a\":3} ", + count: 3, ok: true, + }, + { + name: "a comma inside a string is data", + body: `[{"a":"x,y"},{"a":"z"}]`, + want: " {\"a\":\"x,y\"}\n{\"a\":\"z\"} ", + count: 2, ok: true, + }, + { + name: "brackets inside a string do not move the depth", + body: `[{"a":"]["},{"a":"[[["}]`, + want: " {\"a\":\"][\"}\n{\"a\":\"[[[\"} ", + count: 2, ok: true, + }, + { + name: "an escaped quote does not end the string", + body: `[{"a":"he said \"a,b\""},{"a":"z"}]`, + want: " {\"a\":\"he said \\\"a,b\\\"\"}\n{\"a\":\"z\"} ", + count: 2, ok: true, + }, + { + name: "nested arrays and objects keep their commas", + body: `[{"a":[1,2],"b":{"c":3,"d":4}},{"a":[5]}]`, + want: " {\"a\":[1,2],\"b\":{\"c\":3,\"d\":4}}\n{\"a\":[5]} ", + count: 2, ok: true, + }, + { + // A multi-line array keeps its meaning AND stops carrying layout + // newlines, which is what keeps the record count exact. + name: "a pretty-printed array is one record per line and nothing else", + body: "[\n {\"a\":1},\n {\"a\":2}\n]", + want: " {\"a\":1}\n {\"a\":2} ", + count: 2, ok: true, + }, + { + name: "an empty array holds no records", + body: `[]`, + want: ` `, + count: 0, ok: true, + }, + { + name: "whitespace inside an empty array is still no records", + body: "[\n ]", + want: " ", + count: 0, ok: true, + }, + { + name: "a one-element array is one record", + body: `[{"a":1}]`, + want: ` {"a":1} `, + count: 1, ok: true, + }, + { + name: "scalar elements are still elements", + body: `[1,"x",[2]]`, + want: " 1\n\"x\"\n[2] ", + // They will each be a per-record parse refusal; the framing's job is + // only to keep them separate so the objects around them survive. + count: 3, ok: true, + }, + {name: "a truncated array does not balance", body: `[{"a":1}`, ok: false}, + {name: "a trailing comma cut off does not balance", body: `[{"a":1},`, ok: false}, + {name: "a bare open bracket does not balance", body: `[`, ok: false}, + {name: "a cut-off element does not balance", body: `[{"a":1},{"b`, ok: false}, + {name: "a structural syntax error does not balance", body: `[{"a":1}, {bad]`, ok: false}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + b := []byte(tt.body) + count, ok := reframeArray(b) + require.Equal(t, tt.ok, ok, "balance") + if !tt.ok { + return + } + assert.Equal(t, tt.count, count, "element count") + assert.Equal(t, tt.want, string(b), "rewritten body") + }) + } +} + +// TestReframeArray_DestroysWhatItMustNotSee is the gate, not the scan: these +// bodies are the reason reframeArray runs only behind a leading '['. Running it +// on either would corrupt the record, so the test asserts the damage — if the +// handler ever stops gating, this is the shape of the bug. +func TestReframeArray_DestroysWhatItMustNotSee(t *testing.T) { + t.Parallel() + + obj := []byte(`{"a":1,"b":2}`) + reframeArray(obj) + assert.Equal(t, "{\"a\":1\n\"b\":2}", string(obj), + "a bare object's own commas are at depth 1 — the handler must never send one here") + + ndjson := []byte("{\"a\":1,\"b\":2}\n{\"a\":3,\"b\":4}\n") + reframeArray(ndjson) + assert.NotContains(t, string(ndjson), `{"a":1,"b":2}`, + "an NDJSON body is destroyed too — same reason, same gate") +} + +func TestCellAt(t *testing.T) { + t.Parallel() + const line = `["a, b", "c\"d", 42, null, ["x", "y"], {"k": 1}, "last"]` + for i, want := range []string{`"a, b"`, `"c\"d"`, `42`, `null`, `["x", "y"]`, `{"k": 1}`, `"last"`} { + got, ok := cellAt([]byte(line), i) + require.True(t, ok, "cell %d", i) + assert.Equal(t, want, string(got), "cell %d", i) + } + _, ok := cellAt([]byte(line), 7) + assert.False(t, ok, "past the end") + _, ok = cellAt([]byte(line), -1) + assert.False(t, ok, "before the start") + _, ok = cellAt([]byte(`[`), 0) + assert.False(t, ok, "an unterminated row yields nothing") +} + +// TestEventIDAt: the id is the STORED value, a string cell is JSON-decoded +// because ClickHouse's writer escapes "/" as "\/" — Go's own string-literal +// unquoting refuses that — and a null cell is no id at all. +func TestEventIDAt(t *testing.T) { + t.Parallel() + for _, tt := range []struct { + name string + line string + idx int + want string + ok bool + }{ + {"a quoted string is decoded", `["\/a", "evt-1", 0]`, 1, "evt-1", true}, + {"a slash escape survives", `["\/a\/b", 0]`, 0, "/a/b", true}, + {"a number is its digits", `["x", 18446744073709551615]`, 1, "18446744073709551615", true}, + {"an empty string is no id", `["x", ""]`, 1, "", false}, + // #370: an explicit null is as missing as an absent column — keying on + // its spelling would make every null record one id, and collide with + // the literal id "null". + {"a null cell is no id", `["x", null]`, 1, "", false}, + {"the string null is an id", `["x", "null"]`, 1, "null", true}, + {"a column that is not on the wire is no id", `["x", "y"]`, -1, "", false}, + {"past the end is no id", `["x"]`, 3, "", false}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + got, ok := eventIDAt([]byte(tt.line), tt.idx) + assert.Equal(t, tt.ok, ok) + assert.Equal(t, tt.want, got) + }) + } +} diff --git a/internal/api/ingest_retention_test.go b/internal/api/ingest_retention_test.go index ef65e4b2..d0a6bdd4 100644 --- a/internal/api/ingest_retention_test.go +++ b/internal/api/ingest_retention_test.go @@ -52,7 +52,7 @@ func TestIngest_Dedup_CommitsWithTheAdoptedRetention(t *testing.T) { {Name: "users", Columns: []discovery.Column{{Name: "event_id", Type: "String"}}}, }) dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}) + h := newTestIngestHandler(t, reg, &testutil.MockPublisher{}) h.Dedup = staticDedup(dedup) h.DedupeSettings = (*settings.Store).DedupeFor ingest := func(table, id string) { @@ -83,7 +83,7 @@ func TestIngest_Dedup_CommitsWithTheAdoptedRetention(t *testing.T) { func TestIngest_Dedup_ReloadMidWindowCommitsEachRetention(t *testing.T) { t.Parallel() dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}) + h := newTestIngestHandler(t, testRegistry(t), &testutil.MockPublisher{}) h.Dedup = staticDedup(dedup) var calls atomic.Int32 h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { diff --git a/internal/api/ingest_seams.go b/internal/api/ingest_seams.go deleted file mode 100644 index 210f2ba5..00000000 --- a/internal/api/ingest_seams.go +++ /dev/null @@ -1,82 +0,0 @@ -package api - -import ( - "github.com/Wave-RF/WaveHouse/internal/discovery" -) - -// This file holds the per-record decision points a native type layer will take -// over: schema validation, timestamp canonicalization, and the insert-check -// comparison. Each is an interface with a default implementation that delegates -// to today's code unchanged, so the replacement is a wiring change rather than a -// rewrite of the ingest handler. Nothing here decides anything itself. - -// RecordValidator covers the two schema-driven steps of the record pipeline. -// Validate rejects a record the table's schema cannot accept; CanonicalizeTimestamps -// rewrites DateTime/DateTime64 values to the canonical wire form in place. -// -// CanonicalizeTimestamps returns nothing, matching discovery's function: it is -// deliberately fail-open (#372) — a value it cannot read is passed through for -// ClickHouse to judge, and the row filter is what enforces. Giving it an error -// return would invite a caller to change that. -// -// The two are one interface because they are one contract — "what this schema -// says about this record" — evaluated at two points in prepareRecord that must -// stay apart: the insert-check block sits between them deliberately, so checks -// keep pre-#372 semantics. -type RecordValidator interface { - Validate(schema *discovery.TableSchema, record map[string]any) error - CanonicalizeTimestamps(schema *discovery.TableSchema, record map[string]any) -} - -// discoveryValidator is the default RecordValidator, delegating to -// internal/discovery. -type discoveryValidator struct{} - -func (discoveryValidator) Validate(schema *discovery.TableSchema, record map[string]any) error { - return discovery.Validate(schema, record) -} - -func (discoveryValidator) CanonicalizeTimestamps(schema *discovery.TableSchema, record map[string]any) { - discovery.CanonicalizeTimestamps(schema, record) -} - -// validator returns the handler's RecordValidator, or the default when none is -// wired. Validator is an optional field set after construction, so nil is the -// ordinary case rather than a mistake — and nil must resolve to the enforcing -// default, never to skipping validation. That fail-closed direction is the -// point; see hub.go's rowEvaluator for the same shape on row visibility. -func (h *IngestHandler) validator() RecordValidator { - if h.Validator != nil { - return h.Validator - } - return discoveryValidator{} -} - -// InsertChecker decides whether a record's value satisfies a policy check -// clause. Matches answers the scalar `_eq` form (the required value), InSet the -// `_in` form (set membership). It never sees a record as a whole: the -// auto-injection of a missing check value stays in prepareRecord, where the -// ordering against validation and canonicalization is load-bearing. -type InsertChecker interface { - Matches(actual, required any) bool - InSet(v any, set []any) bool -} - -// canonicalChecker is the default InsertChecker: the canonical-scalar -// comparison in ingest.go, unchanged. -type canonicalChecker struct{} - -func (canonicalChecker) Matches(actual, required any) bool { - return checkValueMatches(actual, required) -} - -func (canonicalChecker) InSet(v any, set []any) bool { return valueInSet(v, set) } - -// checker returns the handler's InsertChecker, or the default when none is -// wired. Nil-safe for the same reason as validator. -func (h *IngestHandler) checker() InsertChecker { - if h.Checker != nil { - return h.Checker - } - return canonicalChecker{} -} diff --git a/internal/api/ingest_seams_test.go b/internal/api/ingest_seams_test.go deleted file mode 100644 index 06335b92..00000000 --- a/internal/api/ingest_seams_test.go +++ /dev/null @@ -1,246 +0,0 @@ -package api - -import ( - "errors" - "net/http" - "net/http/httptest" - "testing" - - "github.com/golang-jwt/jwt/v5" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/Wave-RF/WaveHouse/internal/auth" - "github.com/Wave-RF/WaveHouse/internal/discovery" - "github.com/Wave-RF/WaveHouse/internal/policy" - "github.com/Wave-RF/WaveHouse/internal/testutil" -) - -// recordingValidator observes the two schema-driven steps and can fail -// validation on demand, so a test can prove the handler goes through the seam -// rather than calling discovery directly. -type recordingValidator struct { - validateErr error - validated int - canonicalized int - canonicalizeAs string // non-empty ⇒ stamp this into record["page"] -} - -func (v *recordingValidator) Validate(_ *discovery.TableSchema, _ map[string]any) error { - v.validated++ - return v.validateErr -} - -func (v *recordingValidator) CanonicalizeTimestamps(_ *discovery.TableSchema, record map[string]any) { - v.canonicalized++ - if v.canonicalizeAs != "" { - record["page"] = v.canonicalizeAs - } -} - -// TestIngest_RecordValidatorSeam_IsUsed: a wired RecordValidator replaces both -// steps — its rejection is the record's rejection, and its rewrite is what gets -// published. -func TestIngest_RecordValidatorSeam_IsUsed(t *testing.T) { - t.Parallel() - - t.Run("rejection surfaces as a 400", func(t *testing.T) { - t.Parallel() - pub := &testutil.MockPublisher{} - v := &recordingValidator{validateErr: errors.New("seam says no")} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - h.Validator = v - - w := httptest.NewRecorder() - h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home"}))) - - assert.Equal(t, http.StatusBadRequest, w.Code) - testutil.AssertJSONErrorResponse(t, w) - assert.Contains(t, w.Body.String(), "seam says no") - assert.Equal(t, 1, v.validated) - assert.Zero(t, v.canonicalized, "a rejected record never reaches canonicalization") - assert.Empty(t, pub.Messages) - }) - - t.Run("canonicalization rewrites the published record", func(t *testing.T) { - t.Parallel() - pub := &testutil.MockPublisher{} - v := &recordingValidator{canonicalizeAs: "/rewritten"} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - h.Validator = v - - w := httptest.NewRecorder() - h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home"}))) - - require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) - assert.Equal(t, 1, v.validated) - assert.Equal(t, 1, v.canonicalized) - require.Len(t, pub.Messages, 1) - assert.Contains(t, string(pub.Messages[0].Data), "/rewritten") - }) -} - -// TestIngest_DefaultValidator_WhenUnwired: a handler with no seam wired still -// validates against the schema — the nil case must not read as "allow". -func TestIngest_DefaultValidator_WhenUnwired(t *testing.T) { - t.Parallel() - pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - require.Nil(t, h.Validator) - assert.IsType(t, discoveryValidator{}, h.validator()) - - w := httptest.NewRecorder() - h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"nonexistent_field": 1}))) - assert.Equal(t, http.StatusBadRequest, w.Code) - testutil.AssertJSONErrorResponse(t, w) - assert.Empty(t, pub.Messages) -} - -// alwaysChecker answers every insert check the same way, so a test can tell -// which arm the handler consulted. -type alwaysChecker struct { - matches bool - inSet bool -} - -func (c alwaysChecker) Matches(_, _ any) bool { return c.matches } -func (c alwaysChecker) InSet(_ any, _ []any) bool { return c.inSet } - -// TestIngest_InsertCheckerSeam_IsUsed: a wired InsertChecker decides the check -// clause. The record here would fail the canonical comparison, so a 200 proves -// the seam — not the default — answered. -func TestIngest_InsertCheckerSeam_IsUsed(t *testing.T) { - t.Parallel() - required := "org-allowed" - p := &policy.Policy{Tables: map[string]policy.TablePolicy{ - "clicks": {"viewer": {Insert: &policy.InsertPermissions{ - Check: map[string]policy.Filter{"org_id": {Eq: &required}}, - }}}, - }} - - for _, tt := range []struct { - name string - matches bool - want int - }{ - {"seam admits a value the canonical comparison would reject", true, http.StatusOK}, - {"seam rejects a value the canonical comparison would admit", false, http.StatusForbidden}, - } { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - h.PolicySource = staticPolicy(p) - h.Checker = alwaysChecker{matches: tt.matches} - - value := "org-something-else" - if !tt.matches { - value = required // the default checker would accept this - } - w := httptest.NewRecorder() - h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{"page": "/a", "org_id": value}))) - assert.Equal(t, tt.want, w.Code, "body=%s", w.Body.String()) - if tt.want == http.StatusForbidden { - testutil.AssertJSONErrorResponse(t, w) - } - }) - } -} - -// TestIngest_InsertCheckerSeam_InSet_IsUsed covers the _in arm, which reaches a -// DIFFERENT seam method (InSet, not Matches). Without this, replacing -// h.checker().InSet with the canonical valueInSet leaves the whole suite green — -// the existing _in tests exercise that arm only through the default checker. A -// swapped-in type-aware checker would then take effect for _eq and silently not -// for _in, inside insert-check authorization. -func TestIngest_InsertCheckerSeam_InSet_IsUsed(t *testing.T) { - t.Parallel() - - for _, tt := range []struct { - name string - inSet bool - value string - want int - }{ - {"seam admits a value the canonical comparison would reject", true, "org-z", http.StatusOK}, - {"seam rejects a value the canonical comparison would admit", false, "org-b", http.StatusForbidden}, - } { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - h.PolicySource = checkInStore() - h.Checker = alwaysChecker{inSet: tt.inSet} - - req := ingestRequest(t, "clicks", map[string]any{"page": "/a", "org_id": tt.value}) - ctx := auth.WithRole(req.Context(), "user") - ctx = auth.WithClaims(ctx, jwt.MapClaims{"orgs": []any{"org-a", "org-b"}}) - req = req.WithContext(ctx) - - w := httptest.NewRecorder() - h.Handle(w, withTenant(req)) - assert.Equal(t, tt.want, w.Code, "body=%s", w.Body.String()) - if tt.want == http.StatusForbidden { - testutil.AssertJSONErrorResponse(t, w) - } - }) - } -} - -// TestIngest_DefaultChecker_WhenUnwired: with no seam wired the canonical -// comparison decides, and a mismatched value is still a 403. -func TestIngest_DefaultChecker_WhenUnwired(t *testing.T) { - t.Parallel() - required := "org-allowed" - p := &policy.Policy{Tables: map[string]policy.TablePolicy{ - "clicks": {"viewer": {Insert: &policy.InsertPermissions{ - Check: map[string]policy.Filter{"org_id": {Eq: &required}}, - }}}, - }} - pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - h.PolicySource = staticPolicy(p) - require.Nil(t, h.Checker) - assert.IsType(t, canonicalChecker{}, h.checker()) - - w := httptest.NewRecorder() - h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{"page": "/a", "org_id": "wrong"}))) - assert.Equal(t, http.StatusForbidden, w.Code) - testutil.AssertJSONErrorResponse(t, w) - assert.Empty(t, pub.Messages) -} - -// TestIngest_SeamOrdering_ChecksSitBetweenValidateAndCanonicalize: the two -// RecordValidator calls stay at their current positions with the check-clause -// block between them. Merging them would move check clauses onto canonicalized -// values, silently changing pre-#372 check semantics. -func TestIngest_SeamOrdering_ChecksSitBetweenValidateAndCanonicalize(t *testing.T) { - t.Parallel() - required := "org-allowed" - p := &policy.Policy{Tables: map[string]policy.TablePolicy{ - "clicks": {"viewer": {Insert: &policy.InsertPermissions{ - Check: map[string]policy.Filter{"org_id": {Eq: &required}}, - }}}, - }} - pub := &testutil.MockPublisher{} - v := &recordingValidator{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - h.PolicySource = staticPolicy(p) - h.Validator = v - - // A failing check must land AFTER Validate and BEFORE canonicalization. - w := httptest.NewRecorder() - h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{"page": "/a", "org_id": "wrong"}))) - require.Equal(t, http.StatusForbidden, w.Code) - testutil.AssertJSONErrorResponse(t, w) - assert.Equal(t, 1, v.validated, "validation runs before the check clauses") - assert.Zero(t, v.canonicalized, "canonicalization runs after them, so a failed check never reaches it") -} - -// viewerIngestRequest is ingestRequest with the "viewer" role in context, for -// the policy-gated seam tests. -func viewerIngestRequest(t *testing.T, table string, body map[string]any) *http.Request { - t.Helper() - req := ingestRequest(t, table, body) - return req.WithContext(auth.WithRole(req.Context(), "viewer")) -} diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index 233e3613..12a52fd3 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -26,6 +26,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" + "github.com/Wave-RF/WaveHouse/internal/typelayer" "github.com/golang-jwt/jwt/v5" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -37,15 +38,36 @@ func testRegistry(t testing.TB) *discovery.SchemaRegistry { Name: "clicks", Columns: []discovery.Column{ {Name: "page", Type: "String"}, - {Name: "button", Type: "String", HasDefault: true}, - {Name: "count", Type: "UInt64", HasDefault: true}, - {Name: "event_id", Type: "String", HasDefault: true}, - {Name: "org_id", Type: "String", HasDefault: true}, + {Name: "button", Type: "String", HasDefault: true, DefaultExpression: "''"}, + {Name: "count", Type: "UInt64", HasDefault: true, DefaultExpression: "0"}, + {Name: "event_id", Type: "String", HasDefault: true, DefaultExpression: "''"}, + {Name: "org_id", Type: "String", HasDefault: true, DefaultExpression: "''"}, }, }, }) } +// newTestIngestHandler wires a handler the way production does: over a real +// chtypes engine compiled from the registry's own schemas, bound for +// tenant.Default — testStore's tenant. There is no validation-free mode to test +// against — a nil Types is a 503 by design — so every handler test that reaches +// the body needs one, and skips without the artifact. +func newTestIngestHandler(t testing.TB, reg *discovery.SchemaRegistry, pub mq.Publisher) *IngestHandler { + t.Helper() + h := NewIngestHandler(fixedRegistry(reg), pub) + h.Types = typelayer.TestEngine(t, reg.List()...) + return h +} + +// bindTenants binds reg's tables for each of ids on h's engine, as discovery +// does for every tenant it serves; newTestIngestHandler binds only +// tenant.Default. +func bindTenants(h *IngestHandler, reg *discovery.SchemaRegistry, ids ...tenant.ID) { + for _, id := range ids { + h.Types.Bind(id, typelayer.TestServerVersion, "UTC", reg.List()) + } +} + func ingestRequest(t *testing.T, table string, body any) *http.Request { t.Helper() data, err := json.Marshal(body) @@ -60,7 +82,7 @@ func ingestRequest(t *testing.T, table string, body any) *http.Request { func TestIngest_ValidPayload(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "count": 1}) w := httptest.NewRecorder() @@ -81,7 +103,9 @@ func TestIngest_ValidPayload(t *testing.T) { func TestIngest_PublishesOnTheRequestTenantsTopic(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + reg := testRegistry(t) + h := newTestIngestHandler(t, reg, pub) + bindTenants(h, reg, "acme", "globex") tenants := nestedTenants(t, map[string]string{"acme": fullConfig(100), "globex": fullConfig(200)}) for _, id := range []tenant.ID{"acme", "globex"} { @@ -134,7 +158,7 @@ func TestIngest_MissingTable(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := httptest.NewRequestWithContext( context.Background(), @@ -157,7 +181,7 @@ func TestIngest_MissingTable(t *testing.T) { func TestIngest_UnknownTable(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ingestRequest(t, "nonexistent", map[string]any{"x": 1}) w := httptest.NewRecorder() @@ -171,35 +195,130 @@ func TestIngest_UnknownTable(t *testing.T) { func TestIngest_InvalidJSON(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) r := rawIngestRequest(t, "clicks", "application/json", "not json") w := httptest.NewRecorder() h.Handle(w, withTenant(r)) assert.Equal(t, http.StatusBadRequest, w.Code) - assert.Contains(t, w.Body.String(), "invalid json") + // CONTRACT CHANGE: the message is ClickHouse's own, with its code, because + // nothing in Go reads the body any more. It used to be the flat + // "invalid json" the Go decoder produced. A per-record refusal carries + // exception_code alone: the single-object response renders that record. + msg, code := errorAndCode(t, w) + assert.Contains(t, msg, "expected '{'") + assert.Equal(t, 27, code) + assert.Empty(t, errorClass(t, w), "a per-record refusal carries no class") testutil.AssertJSONErrorResponse(t, w) + assert.Empty(t, pub.Messages) } +// TestIngest_SchemaValidation_UnknownField: a field the table does not have is +// a real ClickHouse rejection (code 117), not a gateway guess. The compile +// profile pins input_format_skip_unknown_fields=0 precisely so this is a +// verdict the caller hears about rather than silent data loss. func TestIngest_SchemaValidation_UnknownField(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "nonexistent_field": 42}) w := httptest.NewRecorder() h.Handle(w, withTenant(req)) assert.Equal(t, http.StatusBadRequest, w.Code) - assert.Contains(t, w.Body.String(), "error") + msg, code := errorAndCode(t, w) + assert.Contains(t, msg, "nonexistent_field") + assert.Equal(t, 117, code) + assert.Empty(t, pub.Messages) +} + +// TestIngest_MissingRequiredColumn_TakesTheDefault records a DELIBERATE +// behaviour change: WaveHouse used to answer 400 "missing required column" for +// a column that is neither nullable nor defaulted. ClickHouse does not — it +// reads an omitted field as the type's default — and the gateway now gives the +// server's answer rather than its own: `{}` into `page String` stores the +// empty string. +func TestIngest_MissingRequiredColumn_TakesTheDefault(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"count": 1}))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, "", publishedData(t, pub)["page"], "the omitted required column takes its type default") +} + +// TestIngest_ComputedColumns_AreNotOnTheWire: the envelope's Columns are the +// columns the exported row actually carries — declaration order minus +// MATERIALIZED, ALIAS and EPHEMERAL. The old envelope used InsertableColumns, +// which counts EPHEMERAL in, so a table with one announced a column the row did +// not have. Supplying one is the server's own 117. +func TestIngest_ComputedColumns_AreNotOnTheWire(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, computedRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/a"}))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + + var evt ingest.EventMessage + require.NoError(t, json.Unmarshal(pub.LastMessage().Data, &evt)) + assert.Equal(t, []string{"page", "country"}, evt.Columns) + assert.JSONEq(t, `["/a", "US"]`, string(evt.Row), "the MATERIALIZED value is the server's to compute") + + w = httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/a", "raw": "x"}))) + require.Equal(t, http.StatusBadRequest, w.Code) + _, code := errorAndCode(t, w) + assert.Equal(t, 117, code, "a record naming an EPHEMERAL column is refused per record, with ClickHouse's code") +} + +// TestIngest_Batch_PerRecordCodes: a batch reports each refused record's own +// ClickHouse code alongside its message, and one bad record does not cost its +// siblings. +func TestIngest_Batch_PerRecordCodes(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + req := rawIngestRequest(t, "clicks", "application/json", + `[{"page":"/a"},{"page":"/b","nope":1},{"page":"/c","count":"x"},{"page":"/d"}]`) + w := httptest.NewRecorder() + h.Handle(w, withTenant(req)) + + require.Equal(t, http.StatusOK, w.Code) + resp := decodeBatchResult(t, w) + assert.Equal(t, 4, resp.Total) + assert.Equal(t, 2, resp.Succeeded) + assert.Equal(t, 2, resp.Failed) + require.Len(t, resp.Results, 4) + assert.True(t, resp.Results[0].Ok) + assert.Equal(t, 117, resp.Results[1].ExceptionCode) + assert.Equal(t, 27, resp.Results[2].ExceptionCode) + assert.True(t, resp.Results[3].Ok) + assert.Len(t, pub.Messages, 2, "the siblings of a refused record still publish") + + // On the wire the per-record code is exception_code; there is no + // per-record string class. + var raw struct { + Results []map[string]any `json:"results"` + } + require.NoError(t, json.Unmarshal(w.Body.Bytes(), &raw)) + assert.InDelta(t, 117, raw.Results[1]["exception_code"], 0) + assert.NotContains(t, raw.Results[1], "code") + assert.NotContains(t, raw.Results[0], "exception_code", "an accepted record carries no code") } func TestIngest_Dedup_FirstTime(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id"} @@ -217,7 +336,7 @@ func TestIngest_Dedup_Duplicate(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id"} @@ -245,7 +364,7 @@ func TestIngest_Dedup_Duplicate(t *testing.T) { func TestIngest_PublishError_503(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull)} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home"}) w := httptest.NewRecorder() @@ -259,7 +378,7 @@ func TestIngest_PublishError_503(t *testing.T) { func TestIngest_PublishUnavailable_503(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: nats: timeout", mq.ErrUnavailable)} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home"}) w := httptest.NewRecorder() @@ -273,7 +392,7 @@ func TestIngest_PublishUnavailable_503(t *testing.T) { func TestIngest_PublishError_500(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{Err: errors.New("some other error")} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home"}) w := httptest.NewRecorder() @@ -287,7 +406,7 @@ func TestIngest_PublishError_500(t *testing.T) { func TestIngest_Policy_Forbidden(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -312,7 +431,7 @@ func TestIngest_Policy_Forbidden(t *testing.T) { func TestIngest_Policy_ColumnDenied(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -329,14 +448,21 @@ func TestIngest_Policy_ColumnDenied(t *testing.T) { w := httptest.NewRecorder() h.Handle(w, withTenant(req)) - assert.Equal(t, http.StatusForbidden, w.Code) - assert.Contains(t, w.Body.String(), "not allowed for insert") + // CONTRACT CHANGE: column policy is answered by compiling the role's own + // schema WITHOUT the denied columns, so the refusal is ClickHouse's + // per-record code 117 — a 400, not the gateway's 403. It no longer confirms + // whether the column exists at all. + assert.Equal(t, http.StatusBadRequest, w.Code) + msg, code := errorAndCode(t, w) + assert.Contains(t, msg, "button") + assert.Equal(t, 117, code) + assert.Empty(t, pub.Messages) } func TestIngest_Policy_CheckClause_Mismatch(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -364,7 +490,7 @@ func TestIngest_Policy_CheckClause_Mismatch(t *testing.T) { func TestIngest_Policy_CheckClause_Match(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -390,15 +516,15 @@ func TestIngest_Policy_CheckClause_Match(t *testing.T) { assert.NotNil(t, pub.LastMessage(), "should have published") } -// TestIngest_Policy_CheckClause_NumericSpellingMatch: the check comparison is -// canonical on both sides (policy.CanonicalScalar), so a numeric claim and a -// numeric insert value match by value even when their JSON spellings differ — -// a claim spelled 1.0 accepts an inserted 1. The former string-form comparison -// ("1.0" != "1") rejected exactly this insert. +// TestIngest_Policy_CheckClause_NumericSpellingMatch: a numeric CLAIM still +// matches a numeric insert value whose JSON spelling differs — a claim spelled +// 1.0 accepts an inserted 1. Policy canonicalizes the claim to "1" on the way +// in, and the check then binds it as a String parameter that ClickHouse +// compares against the stored UInt64. Nothing in Go compares the two values. func TestIngest_Policy_CheckClause_NumericSpellingMatch(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) countTemplate := "{{ jwt.max_count }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -423,26 +549,41 @@ func TestIngest_Policy_CheckClause_NumericSpellingMatch(t *testing.T) { assert.NotNil(t, pub.LastMessage(), "should have published") } -// TestIngest_Policy_CheckClause_StaticNumericSpelling: a check value with no -// placeholder carries no JSON type, so the comparison accepts either reading -// of it — a static `_eq: "1.0"` accepts an inserted number 1 (its canonical -// numeric reading) and an inserted string "1.0" (its spelling, the pre-PR -// behavior) alike. Without the numeric reading, the payload side is canonical -// ("1") while the static side keeps its raw spelling ("1.0") and the check -// rejects every numeric insert it was written to allow. +// TestIngest_Policy_CheckClause_StaticNumericSpelling is a DOCUMENTED CONTRACT +// CHANGE. A placeholder-free check value used to carry a second, numeric +// reading in Go — a typed marker on the resolved clause plus a canonical +// numeric re-render of the literal, both since deleted — so a static +// `_eq: "1.0"` admitted an inserted number 1. The check is now ClickHouse's own +// comparison, and on an integer column the claim goes through the strict cast +// (chsql.StrictInt): "1.0" is not the canonical spelling of a UInt64, so it +// matches nothing and the record is refused as a failed check (403) — the +// same answer as a claim past the column's range. +// +// The value is also what would have been injected into a record that omitted +// the column, and `count UInt64 DEFAULT '1.0'` does not compile (code 6, +// measured). The handler retries the shape without its defaults rather than +// answering 503, so the request still gets a per-record verdict. +// +// The operator fix is to write the literal the column can read (`_eq: "1"`); +// the covering case below pins that it still works. func TestIngest_Policy_CheckClause_StaticNumericSpelling(t *testing.T) { t.Parallel() for _, tt := range []struct { - name string - body any + name string + body any + want int + checkFailed bool }{ - {"numeric reading", 1}, - {"literal spelling", "1.0"}, + {"numeric reading", 1, http.StatusForbidden, true}, + // ClickHouse refuses the record before the check is ever consulted: the + // column is a UInt64 and the string "1.0" is not one. Parse errors + // precede check errors now (TestIngest_ParseErrorsPrecedeCheckErrors). + {"literal spelling", "1.0", http.StatusBadRequest, false}, } { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) staticCount := "1.0" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -462,18 +603,88 @@ func TestIngest_Policy_CheckClause_StaticNumericSpelling(t *testing.T) { w := httptest.NewRecorder() h.Handle(w, withTenant(req)) - assert.Equal(t, http.StatusOK, w.Code) - assert.NotNil(t, pub.LastMessage(), "should have published") + assert.Equal(t, tt.want, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, tt.checkFailed, strings.Contains(w.Body.String(), "check failed"), + "the claim that does not fit is the check saying no; the parse refusal comes first") + assert.Empty(t, pub.Messages) + }) + } + + // The covering case: a literal the column CAN read still admits the record, + // so the change above is about the spelling, not about static checks. + t.Run("a readable literal still admits the record", func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + staticCount := "1" + h.PolicySource = staticPolicy(&policy.Policy{ + Tables: map[string]policy.TablePolicy{ + "clicks": {"user": {Insert: &policy.InsertPermissions{Check: map[string]policy.Filter{ + "count": {Eq: &staticCount}, + }}}}, + }, + }) + req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "count": 1}) + req = req.WithContext(auth.WithRole(req.Context(), "user")) + w := httptest.NewRecorder() + h.Handle(w, withTenant(req)) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Len(t, pub.Messages, 1) + }) +} + +// TestIngest_Policy_CheckClause_IntegerClaimThatDoesNotFit_IsRefused: an +// integer claim is compared through the strict round-trip cast, so a claim the +// column cannot hold (2^64+5 on a UInt64) refuses every record — the one that +// omits the column (its injected DEFAULT does not compile, so it takes the +// table's own and the filter refuses it) and the one whose value a plain String +// binding would have wrapped the claim onto (5). A claim that fits admits +// exactly the records that carry it. +func TestIngest_Policy_CheckClause_IntegerClaimThatDoesNotFit_IsRefused(t *testing.T) { + t.Parallel() + countTemplate := "{{ jwt.max_count }}" + p := &policy.Policy{Tables: map[string]policy.TablePolicy{ + "clicks": {"user": {Insert: &policy.InsertPermissions{Check: map[string]policy.Filter{ + "count": {Eq: &countTemplate}, + }}}}, + }} + for _, tt := range []struct { + name string + claim string + ok []bool + }{ + {"a claim past the column's range", "18446744073709551621", []bool{false, false, false}}, + {"a claim that fits", "5", []bool{true, true, false}}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + h.PolicySource = staticPolicy(p) + req := ndjsonRequest(t, "clicks", `{"page":"/a"}`, `{"page":"/b","count":5}`, `{"page":"/c","count":6}`) + ctx := auth.WithClaims(auth.WithRole(req.Context(), "user"), jwt.MapClaims{"max_count": tt.claim}) + w := httptest.NewRecorder() + h.Handle(w, withTenant(req.WithContext(ctx))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + for i, want := range tt.ok { + r := resultAt(t, resp, i+1) + assert.Equal(t, want, r.Ok, "record %d: %+v", i+1, r) + if !want { + assert.Contains(t, r.Error, `check failed for column "count"`, "record %d", i+1) + } + } }) } } -// TestIngest_Policy_CheckClause_StringClaimStrictEquality: the numeric -// reading is reserved for placeholder-free literals (policy.LiteralValue). A -// claim-derived required value keeps strict canonical equality, so a writer -// whose claim is the STRING "1e3" cannot insert the number 1000 — its id is -// the three-character text, and accepting the numeric reading would let it -// store a row under the tenant whose String id is "1000". +// TestIngest_Policy_CheckClause_StringClaimStrictEquality: a writer whose claim +// is the STRING "1e3" cannot insert the number 1000 — its id is the +// three-character text, and a numeric reading would let it store a row under +// the tenant whose String id is "1000". ClickHouse enforces it now: the column +// is a String, so the comparison is between strings and no numeric reading +// exists to slip through. func TestIngest_Policy_CheckClause_StringClaimStrictEquality(t *testing.T) { t.Parallel() for _, tt := range []struct { @@ -487,7 +698,7 @@ func TestIngest_Policy_CheckClause_StringClaimStrictEquality(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -512,15 +723,25 @@ func TestIngest_Policy_CheckClause_StringClaimStrictEquality(t *testing.T) { } } -// TestIngest_Policy_CheckClause_NullValue_FailsClosed: the _eq twin of the -// _in null test. With the claim absent the required value is "", and -// CanonicalScalar(nil) also renders "" — only the hasForm guard separates -// them, so without it an explicit null in the payload would satisfy the check -// (pre-canonicalization, fmt.Sprint(nil) gave "" and this was impossible). -func TestIngest_Policy_CheckClause_NullValue_FailsClosed(t *testing.T) { +// TestIngest_Policy_CheckClause_NullValue_StoredValueIsWhatIsChecked is a +// DOCUMENTED CONTRACT CHANGE, and the reason is worth stating precisely because +// it reads as a loosening. +// +// The check is now evaluated against the row ClickHouse WOULD STORE, not against +// the caller's spelling. `input_format_null_as_default=1` — the setting the real +// INSERT already pins — turns an explicit JSON null on a non-nullable column +// into that column's default, so `org_id: null` stores "". The required value +// here is also "": policy deliberately resolves an unresolvable check claim to +// the empty string and auto-injects it (#463), so a record OMITTING org_id has +// always been accepted and stored "". The two cases now agree, where the Go-side +// rule ("null has no canonical form, so it matches nothing") made them differ. +// +// Nothing is admitted that the stored row does not satisfy — which is the +// property the check clause is for. +func TestIngest_Policy_CheckClause_NullValue_StoredValueIsWhatIsChecked(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -540,16 +761,47 @@ func TestIngest_Policy_CheckClause_NullValue_FailsClosed(t *testing.T) { w := httptest.NewRecorder() h.Handle(w, withTenant(req)) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, "", publishedData(t, pub)["org_id"], "the stored value is what satisfied the check") + + // With a claim that DOES resolve, an explicit null takes the INJECTED value + // rather than the table's own default: null_as_default resolves it against + // the ROLE's compiled schema, whose DEFAULT is the claim. A null on a checked + // column therefore behaves exactly like omitting it, and can never carry + // another tenant's value — the property that matters. + pub2 := &testutil.MockPublisher{} + h2 := newTestIngestHandler(t, testRegistry(t), pub2) + h2.PolicySource = h.PolicySource + req = ingestRequest(t, "clicks", map[string]any{"page": "/home", "org_id": nil}) + ctx = auth.WithRole(req.Context(), "user") + ctx = auth.WithClaims(ctx, jwt.MapClaims{"org_id": "real-org"}) + w = httptest.NewRecorder() + h2.Handle(w, withTenant(req.WithContext(ctx))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, "real-org", publishedData(t, pub2)["org_id"]) + + // A null is not a way past the check: a value that is present and wrong is + // still refused. + pub3 := &testutil.MockPublisher{} + h3 := newTestIngestHandler(t, testRegistry(t), pub3) + h3.PolicySource = h.PolicySource + req = ingestRequest(t, "clicks", map[string]any{"page": "/home", "org_id": "someone-else"}) + ctx = auth.WithRole(req.Context(), "user") + ctx = auth.WithClaims(ctx, jwt.MapClaims{"org_id": "real-org"}) + w = httptest.NewRecorder() + h3.Handle(w, withTenant(req.WithContext(ctx))) + assert.Equal(t, http.StatusForbidden, w.Code) assert.Contains(t, w.Body.String(), "check failed") - assert.Nil(t, pub.LastMessage(), "a null payload value must not satisfy an unresolvable check") + assert.Empty(t, pub3.Messages) testutil.AssertJSONErrorResponse(t, w) } func TestIngest_Policy_CheckClause_AutoInject(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -595,7 +847,7 @@ func checkInStore() PolicySource { func TestIngest_Policy_CheckIn_InSet(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = checkInStore() // org_id is one of the token's allowed orgs — should pass. @@ -614,7 +866,7 @@ func TestIngest_Policy_CheckIn_InSet(t *testing.T) { func TestIngest_Policy_CheckIn_NotInSet(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = checkInStore() // org_id is NOT one of the token's allowed orgs — forging another tenant's row. @@ -631,15 +883,22 @@ func TestIngest_Policy_CheckIn_NotInSet(t *testing.T) { testutil.AssertJSONErrorResponse(t, w) } -// TestIngest_Policy_CheckIn_NullValue_FailsClosed: an explicit null passes -// schema validation for a defaulted column, but it has no canonical scalar -// form — so it is a member of NO set, even one carrying the empty string "" -// (the regression shape: an unguarded canonicalization would render null as -// "" and match an empty-string member). -func TestIngest_Policy_CheckIn_NullValue_FailsClosed(t *testing.T) { +// TestIngest_Policy_CheckIn_NullValue_ChecksTheStoredValue is a DOCUMENTED +// CONTRACT CHANGE, the _in twin of the _eq one above. An explicit null on a +// non-nullable column stores that column's default (input_format_null_as_default +// =1, the setting the real INSERT pins), and an _in check has no value to +// inject, so the stored "" is what the filter tests. A token whose allowed set +// LISTS "" therefore admits it — the row it stores really is one the token +// authorises. The old Go-side rule said null has no canonical form and matched +// nothing. +// +// The second half is the part that must not move: an allowed set WITHOUT "" is +// still a refusal, so this is not a way to write a row under a tenant the token +// does not carry. +func TestIngest_Policy_CheckIn_NullValue_ChecksTheStoredValue(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = checkInStore() req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "org_id": nil}) @@ -650,15 +909,28 @@ func TestIngest_Policy_CheckIn_NullValue_FailsClosed(t *testing.T) { w := httptest.NewRecorder() h.Handle(w, withTenant(req)) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, "", publishedData(t, pub)["org_id"], `the stored "" is a member of the token's own set`) + + pub2 := &testutil.MockPublisher{} + h2 := newTestIngestHandler(t, testRegistry(t), pub2) + h2.PolicySource = checkInStore() + req = ingestRequest(t, "clicks", map[string]any{"page": "/home", "org_id": nil}) + ctx = auth.WithRole(req.Context(), "user") + ctx = auth.WithClaims(ctx, jwt.MapClaims{"orgs": []any{"org-a", "org-b"}}) + w = httptest.NewRecorder() + h2.Handle(w, withTenant(req.WithContext(ctx))) + assert.Equal(t, http.StatusForbidden, w.Code) assert.Contains(t, w.Body.String(), "check failed") + assert.Empty(t, pub2.Messages, `a set without "" still refuses the null`) testutil.AssertJSONErrorResponse(t, w) } func TestIngest_Policy_CheckIn_Absent_FailsClosed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = checkInStore() // org_id omitted — unlike _eq there's no single value to auto-inject, so the @@ -678,14 +950,14 @@ func TestIngest_Policy_CheckIn_Absent_FailsClosed(t *testing.T) { // TestIngest_Policy_CheckIn_AbsentClaim_FailsClosed locks the typed-nil []any // path behind an _in check: when the claim itself is absent, resolveInValues -// returns a typed-nil []any, which must still assert as []any in prepareRecord -// (entering the membership branch) so the column is rejected — never treated as a -// scalar _eq value and auto-injected. The sibling _Absent test omits the column +// returns a typed-nil []any, which must still assert as []any in insertShape +// (becoming an `in` predicate that matches nothing) so the record is rejected — +// never treated as a scalar _eq value and auto-injected. The sibling _Absent test omits the column // with the claim present; this one drops the claim too. Guards #224 fail-closed. func TestIngest_Policy_CheckIn_AbsentClaim_FailsClosed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = checkInStore() // The `orgs` claim is absent entirely, so the _in set resolves to a typed-nil @@ -719,7 +991,9 @@ func TestIngest_DedupIsTheTenants(t *testing.T) { require.NoError(t, stores.For(id).Apply(true)) } pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + reg := testRegistry(t) + h := newTestIngestHandler(t, reg, pub) + bindTenants(h, reg, "acme", "globex") h.Dedup = func(s *settings.Store) dedupe.Deduplicator { return stores.For(s.Tenant()) } h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id"} @@ -744,7 +1018,7 @@ func TestIngest_Dedup_MissingIDField(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id"} @@ -765,7 +1039,7 @@ func TestIngest_Dedup_MissingIDField(t *testing.T) { func TestIngest_Dedup_RequireID_Rejects(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(testutil.NewMockDeduplicator()) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} @@ -789,7 +1063,7 @@ func TestIngest_Dedup_RequireID_Rejects(t *testing.T) { func TestIngest_NDJSON_RequireID_Rejects(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(testutil.NewMockDeduplicator()) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} @@ -818,7 +1092,7 @@ func TestIngest_NDJSON_RequireID_Rejects(t *testing.T) { func TestIngest_Policy_DenyColumns(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -834,14 +1108,28 @@ func TestIngest_Policy_DenyColumns(t *testing.T) { w := httptest.NewRecorder() h.Handle(w, withTenant(req)) - assert.Equal(t, http.StatusForbidden, w.Code) - assert.Contains(t, w.Body.String(), "not allowed for insert") + // CONTRACT CHANGE, as for AllowColumns: a denied column is absent from the + // role's compiled schema, so naming it is ClickHouse's code 117. + assert.Equal(t, http.StatusBadRequest, w.Code) + msg, code := errorAndCode(t, w) + assert.Contains(t, msg, "count") + assert.Equal(t, 117, code) + assert.Empty(t, pub.Messages) + + // And the deny is real, not an artifact of the message: the same record + // without the denied column is accepted. + req = ingestRequest(t, "clicks", map[string]any{"page": "/home"}) + req = req.WithContext(auth.WithRole(req.Context(), "writer")) + w = httptest.NewRecorder() + h.Handle(w, withTenant(req)) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Len(t, pub.Messages, 1) } func TestIngest_AdminRole_NoPolicy(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": {}, @@ -905,7 +1193,7 @@ func resultAt(t *testing.T, resp batchResult, index int) recordResult { func TestIngest_NDJSON_AllValid(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a", "count": 1}), @@ -933,7 +1221,7 @@ func TestIngest_NDJSON_AllValid(t *testing.T) { func TestIngest_NDJSON_PartialFailure_Validation(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"}), @@ -959,7 +1247,7 @@ func TestIngest_NDJSON_PartialFailure_Validation(t *testing.T) { func TestIngest_NDJSON_MalformedLine(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"}), @@ -976,7 +1264,12 @@ func TestIngest_NDJSON_MalformedLine(t *testing.T) { assert.Equal(t, 1, resp.Failed) require.Len(t, resp.Results, 3) assert.True(t, resultAt(t, resp, 1).Ok) - assert.Contains(t, resultAt(t, resp, 2).Error, "invalid json") + // The message is ClickHouse's own parse refusal now, not Go's "invalid json", + // and the code is whichever one its reader raised (26 for a quoted string it + // cannot finish, 27 for a value it cannot read) — the point is that a code is + // attributed at all. + assert.NotZero(t, resultAt(t, resp, 2).ExceptionCode) + assert.NotEmpty(t, resultAt(t, resp, 2).Error) assert.True(t, resultAt(t, resp, 3).Ok) assert.Len(t, pub.Messages, 2) } @@ -984,7 +1277,7 @@ func TestIngest_NDJSON_MalformedLine(t *testing.T) { func TestIngest_NDJSON_BlankLinesSkipped(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) // Leading, interior, and whitespace-only lines are all skipped; only real // records are counted. @@ -1021,7 +1314,7 @@ func TestIngest_NDJSON_EmptyBody(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ndjsonRequest(t, "clicks", tt.lines...) w := httptest.NewRecorder() @@ -1039,7 +1332,7 @@ func TestIngest_NDJSON_Dedup(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id"} @@ -1070,7 +1363,7 @@ func TestIngest_NDJSON_Backpressure_503(t *testing.T) { // Publisher rejects every publish with the backpressure sentinel; the first // valid record aborts the whole batch with 503 + Retry-After. pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull)} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"}), @@ -1087,7 +1380,7 @@ func TestIngest_NDJSON_Backpressure_503(t *testing.T) { func TestIngest_NDJSON_Unavailable_503(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: nats: no responders", mq.ErrUnavailable)} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"})) w := httptest.NewRecorder() @@ -1101,7 +1394,7 @@ func TestIngest_NDJSON_Unavailable_503(t *testing.T) { func TestIngest_NDJSON_PublishError_500(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{Err: errors.New("some other error")} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"})) w := httptest.NewRecorder() @@ -1115,7 +1408,7 @@ func TestIngest_NDJSON_PublishError_500(t *testing.T) { func TestIngest_NDJSON_Policy_ColumnDenied_PerLine(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -1141,14 +1434,16 @@ func TestIngest_NDJSON_Policy_ColumnDenied_PerLine(t *testing.T) { assert.Equal(t, 1, resp.Failed) require.Len(t, resp.Results, 2) assert.True(t, resultAt(t, resp, 1).Ok) - assert.Contains(t, resultAt(t, resp, 2).Error, "not allowed for insert") + // CONTRACT CHANGE: the denied column is ClickHouse's code 117. + assert.Contains(t, resultAt(t, resp, 2).Error, "button") + assert.Equal(t, 117, resultAt(t, resp, 2).ExceptionCode) assert.Len(t, pub.Messages, 1) } func TestIngest_NDJSON_Policy_TableForbidden(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -1175,7 +1470,7 @@ func TestIngest_NDJSON_Policy_TableForbidden(t *testing.T) { func TestIngest_NDJSON_Policy_CheckClause_PerLineAndAutoInject(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -1215,7 +1510,7 @@ func TestIngest_NDJSON_Policy_CheckClause_PerLineAndAutoInject(t *testing.T) { func TestIngest_NDJSON_ContentTypeWithCharset(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"})) req.Header.Set("Content-Type", "application/x-ndjson; charset=utf-8") @@ -1232,7 +1527,7 @@ func TestIngest_NDJSON_ContentTypeWithCharset(t *testing.T) { func TestIngest_NDJSON_ErrorsTruncated(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) const total = maxReportedResults + 50 lines := make([]string, total) @@ -1263,7 +1558,7 @@ func TestIngest_NDJSON_ErrorsTruncated(t *testing.T) { // spell the whole list out, and always need editing: api.md's body/Content-Type // table, architecture.md's "the four NDJSON spellings" count, and the ingest // entry in CHANGELOG.md. -const wantAcceptedTypes = "application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines" +const wantAcceptedTypes = "application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent" // TestAcceptedTypesAreAllResolvable pins that the advertised list never grows // beyond what the resolver accepts — an entry added to acceptedContentTypes but @@ -1341,8 +1636,35 @@ func TestIngestFormat(t *testing.T) { // Tracked in #563. {ct: `application/json; profile="a,b"; charset`, wantErr: true}, {ct: "APPLICATION/JSON", want: FormatJSON}, + {ct: "text/csv", want: FormatCSV}, + {ct: "text/csv; charset=utf-8", want: FormatCSV}, + {ct: "text/tab-separated-values", want: FormatTSV}, + // RFC 4180 §3's header parameter is the one parameter that decides a + // format, and only for the CSV/TSV pair: present, absent and no parameter + // are three different readings. The value is matched case-insensitively. + {ct: "text/csv; header=present", want: FormatCSVWithNames}, + {ct: "text/csv; charset=utf-8; header=present", want: FormatCSVWithNames}, + {ct: "text/csv; header=PRESENT", want: FormatCSVWithNames}, + {ct: "text/csv; header=absent", want: FormatCSVPositional}, + {ct: "text/csv; charset=utf-8; header=ABSENT", want: FormatCSVPositional}, + {ct: "text/tab-separated-values; header=present", want: FormatTSVWithNames}, + {ct: "text/tab-separated-values; header=absent", want: FormatTSVPositional}, + {ct: "application/json; header=present", want: FormatJSON}, + // A value that is neither is refused rather than guessed at, and so is a + // line whose parameters did not parse when it mentions a header: reading + // it as absent would ingest a declared header line as data. + {ct: "text/csv; header=yes", wantErr: true}, + {ct: "text/csv; header=", wantErr: true}, + {ct: "text/csv; charset; header=present", wantErr: true}, + {ct: "text/csv; header=present; header=absent", wantErr: true}, + {ct: "text/csv; charset", want: FormatCSV}, {ct: "text/plain", wantErr: true}, - {ct: "text/csv", wantErr: true}, + // Near misses for the positional pair, for the same reason as the JSON + // ones below: an exact-match lookup rewritten as a prefix test would + // start ingesting these with the suite green. + {ct: "text/csv2", wantErr: true}, + {ct: "text/tab-separated-value", wantErr: true}, + {ct: "text/tsv", wantErr: true}, {ct: "", wantErr: true}, {ct: " ", wantErr: true}, {ct: "???not-a-media-type", wantErr: true}, @@ -1409,14 +1731,14 @@ func TestIngest_UndeclaredOrUnsupportedContentType_415(t *testing.T) { }{ {"no content-type", "", "no Content-Type: "}, {"text/plain", "text/plain", `Content-Type "text/plain": `}, - {"text/csv", "text/csv", `Content-Type "text/csv": `}, + {"application/xml", "application/xml", `Content-Type "application/xml": `}, {"malformed media type", "???not-a-media-type", `Content-Type "???not-a-media-type": `}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", tt.ct, `{"page":"/a"}`))) @@ -1451,6 +1773,35 @@ func jsonErrorMessage(t *testing.T, w *httptest.ResponseRecorder) string { return body.Error } +// errorAndCode returns the decoded "error" message and ClickHouse's +// "exception_code", which is absent (0) unless the server's own parser is what +// refused. +func errorAndCode(t *testing.T, w *httptest.ResponseRecorder) (string, int) { + t.Helper() + var body struct { + Error string `json:"error"` + ExceptionCode int `json:"exception_code"` + } + require.NoError(t, json.Unmarshal(w.Body.Bytes(), &body)) + return body.Error, body.ExceptionCode +} + +// errorClass returns the error body's string "code" — the failure class, "" +// when the body carries none. +func errorClass(t *testing.T, w *httptest.ResponseRecorder) string { + t.Helper() + var body struct { + Code any `json:"code"` + } + require.NoError(t, json.Unmarshal(w.Body.Bytes(), &body)) + if body.Code == nil { + return "" + } + s, ok := body.Code.(string) + require.True(t, ok, "code must be the string class, got %T", body.Code) + return s +} + // TestIngest_ContentTypeRefusalBeatsEmptyBody: the PR's headline ordering claim // — "checked before the body is parsed" — is what lets a caller trust that a 415 // describes their header and not their payload. Nothing pinned it: every 415 case @@ -1466,7 +1817,7 @@ func TestIngest_ContentTypeRefusalBeatsEmptyBody(t *testing.T) { t.Run(name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", ct, ""))) @@ -1482,18 +1833,41 @@ func TestIngest_ContentTypeRefusalBeatsEmptyBody(t *testing.T) { // TestIngest_DeclaredNDJSON_ArrayBodyIsNotReframed: the header is authoritative. // A JSON array sent as NDJSON is read as NDJSON — one line, not a JSON object — // so it fails as a per-record error instead of silently being re-read as a batch. +// TestIngest_DeclaredNDJSON_ArrayBodyIsNotReframed: the declared format is still +// authoritative — a declared-NDJSON body is never re-read as the JSON family, so +// the depth-1 comma rewrite (which is what makes a compact array salvageable per +// record) does not run on it. +// +// CONTRACT CHANGE: it used to be one unparseable NDJSON line, reported as a +// single per-record failure. ClickHouse's own JSONEachRow reader takes the +// surrounding brackets in its stride, so both objects now ingest and the batch +// reports two records. Nothing is silently dropped either way; what changed is +// that the mis-declaration now costs nothing instead of the whole body. func TestIngest_DeclaredNDJSON_ArrayBodyIsNotReframed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "application/x-ndjson", `[{"page":"/a"},{"page":"/b"}]`))) assert.Equal(t, http.StatusOK, w.Code) resp := decodeBatchResult(t, w) - assert.Equal(t, 1, resp.Total, "the array is one NDJSON line, not two records") - assert.Equal(t, 1, resp.Failed) - assert.Empty(t, pub.Messages) + assert.Equal(t, 2, resp.Total, "ClickHouse's reader frames the array's elements") + assert.Equal(t, 2, resp.Succeeded) + assert.Len(t, pub.Messages, 2) + + // The rewrite really is off for this declaration: a compact array with one + // bad record loses the whole batch here, which is exactly the cliff the + // rewrite exists to remove for a declared-JSON body (see + // TestIngest_JSONArray_CompactWithOneBadRecord). + pub2 := &testutil.MockPublisher{} + h2 := newTestIngestHandler(t, testRegistry(t), pub2) + w = httptest.NewRecorder() + h2.Handle(w, withTenant(rawIngestRequest(t, "clicks", "application/x-ndjson", + `[{"page":"/a"},{"page":"/b","nope":1},{"page":"/c"}]`))) + require.Equal(t, http.StatusOK, w.Code) + assert.Zero(t, decodeBatchResult(t, w).Succeeded, "the compact array is all-or-nothing without the rewrite") + assert.Empty(t, pub2.Messages) } // ── Multi-format ingest (JSON array, arity sniffing, body cap) ───────────── @@ -1532,7 +1906,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("disagreeing declarations are refused", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "application/json", ndjson) req.Header.Add("Content-Type", "application/x-ndjson") @@ -1557,7 +1931,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("a supported and an unsupported declaration are refused", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "application/json", ndjson) req.Header.Add("Content-Type", "text/csv") @@ -1576,7 +1950,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("different spellings of the same format are accepted", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "application/x-ndjson", ndjson) req.Header.Add("Content-Type", "application/ndjson; charset=utf-8") @@ -1606,7 +1980,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run(name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", ct, ndjson))) @@ -1651,7 +2025,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run(name, func(t *testing.T) { t.Parallel() wJ := httptest.NewRecorder() - NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}). + newTestIngestHandler(t, testRegistry(t), &testutil.MockPublisher{}). Handle(wJ, withTenant(rawIngestRequest(t, "clicks", tc.joined, `{"page":"/a"}`))) assert.Equal(t, tc.wJoined, wJ.Code, "joined") @@ -1660,7 +2034,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { req.Header.Add("Content-Type", v) } wR := httptest.NewRecorder() - NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}). + newTestIngestHandler(t, testRegistry(t), &testutil.MockPublisher{}). Handle(wR, withTenant(req)) assert.Equal(t, tc.wRepeat, wR.Code, "repeated") }) @@ -1670,7 +2044,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("a quoted comma does not split a declaration", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", `application/json; profile="a,b"`, `{"page":"/a"}`))) @@ -1681,7 +2055,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("a third line that disagrees is refused", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "application/json", ndjson) req.Header.Add("Content-Type", "application/json") req.Header.Add("Content-Type", "application/x-ndjson") @@ -1703,7 +2077,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Parallel() for _, first := range []bool{false, true} { pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "", ndjson) if first { req.Header.Add("Content-Type", empty) @@ -1729,8 +2103,8 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("two unsupported lines name both", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - req := rawIngestRequest(t, "clicks", "text/csv", ndjson) + h := newTestIngestHandler(t, testRegistry(t), pub) + req := rawIngestRequest(t, "clicks", "application/xml", ndjson) req.Header.Add("Content-Type", "text/plain") w := httptest.NewRecorder() @@ -1739,7 +2113,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { assert.Equal(t, http.StatusUnsupportedMediaType, w.Code) testutil.AssertJSONErrorResponse(t, w) msg := jsonErrorMessage(t, w) - assert.Contains(t, msg, `"text/csv"`) + assert.Contains(t, msg, `"application/xml"`) assert.Contains(t, msg, `"text/plain"`, "a declaration the caller sent must not vanish from the message") assert.NotContains(t, msg, "conflicting", "agreeing-but-unsupported is not a conflict") assert.Empty(t, pub.Messages) @@ -1748,7 +2122,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("an identical declaration repeated is not ambiguous", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "application/x-ndjson", ndjson) req.Header.Add("Content-Type", "application/x-ndjson") @@ -1763,7 +2137,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { func TestIngest_JSONArray_AllValid(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) // A JSON array declared as application/json is read as a batch — the body's // first byte picks arity within the family ingestRequest declares. @@ -1788,7 +2162,7 @@ func TestIngest_JSONArray_AllValid(t *testing.T) { func TestIngest_JSONArray_SingleElement(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) // A one-element array is still a batch (returns the results envelope, not // the single-object {"ok":true}). @@ -1808,7 +2182,7 @@ func TestIngest_JSONArray_SingleElement(t *testing.T) { func TestIngest_JSONArray_PartialValidationFailure(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := ingestRequest(t, "clicks", []map[string]any{ {"page": "/a"}, @@ -1833,11 +2207,11 @@ func TestIngest_JSONArray_PartialValidationFailure(t *testing.T) { func TestIngest_JSONArray_ScalarElements(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) // Non-object elements (number, string, nested array) are wrong-typed: the - // decoder stays in sync, so each is a per-record error and the objects - // around them still ingest. + // framing keeps each on its own line, so each is ClickHouse's per-record + // refusal and the objects around them still ingest. req := ingestRequest(t, "clicks", []any{ map[string]any{"page": "/a"}, 5, @@ -1864,11 +2238,11 @@ func TestIngest_JSONArray_ScalarElements(t *testing.T) { func TestIngest_JSONArray_SyntaxError_Fatal(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) - // A structural syntax error desyncs the decoder — the whole request fails - // (400), unlike a per-element type error. The leading good element is still - // in the open window, which is dropped unpublished. + // A structural syntax error leaves the brackets unbalanced — the whole + // request fails (400), unlike a per-element refusal. Nothing is judged, so + // the leading good element is not published either. req := rawIngestRequest(t, "clicks", "application/json", `[{"page":"/a"}, {bad]`) w := httptest.NewRecorder() h.Handle(w, withTenant(req)) @@ -1898,7 +2272,7 @@ func TestIngest_JSONArray_Truncated_Fatal(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "application/json", tt.body) w := httptest.NewRecorder() @@ -1914,7 +2288,7 @@ func TestIngest_JSONArray_Truncated_Fatal(t *testing.T) { func TestIngest_JSONArray_Empty(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) // An explicit empty array is a valid, record-less batch → 200 with no rows. req := rawIngestRequest(t, "clicks", "application/json", `[]`) @@ -1931,7 +2305,7 @@ func TestIngest_JSONArray_Empty(t *testing.T) { func TestIngest_SingleObject_PrettyPrinted(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) // A multi-line (pretty-printed) single object must not be mistaken for // NDJSON — it's one record on the single-object path. @@ -1949,7 +2323,7 @@ func TestIngest_SingleObject_PrettyPrinted(t *testing.T) { func TestIngest_DeclaredJSON_ConcatenatedObjects_FirstOnly(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) // Two concatenated objects declared as application/json take the // single-object path and ingest only the first (matching the historical @@ -1982,7 +2356,7 @@ func TestIngest_LeadingWhitespace_Sniff(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "application/json", tt.body) w := httptest.NewRecorder() @@ -2020,7 +2394,7 @@ func TestIngest_EmptyBody(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", tt.contentType, tt.body) w := httptest.NewRecorder() @@ -2051,7 +2425,7 @@ func TestIngest_BodyReadFailure_400(t *testing.T) { t.Run(ct, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := httptest.NewRequestWithContext(context.Background(), http.MethodPost, "/v1/ingest?table=clicks", iotest.ErrReader(errors.New("connection reset by peer"))) @@ -2103,7 +2477,7 @@ func TestIngest_BodyCap_413(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.maxRequestBytes = tt.cap // below the body req := rawIngestRequest(t, "clicks", tt.ct, tt.body) @@ -2137,11 +2511,11 @@ func TestIngest_ContentTypeResolvesBeforeTheBodyIsRead(t *testing.T) { t.Run("a body that cannot be read at all", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := httptest.NewRequestWithContext(context.Background(), http.MethodPost, "/v1/ingest?table=clicks", iotest.ErrReader(errors.New("connection reset by peer"))) - req.Header.Set("Content-Type", "text/csv") + req.Header.Set("Content-Type", "application/xml") w := httptest.NewRecorder() h.Handle(w, withTenant(req)) @@ -2155,10 +2529,10 @@ func TestIngest_ContentTypeResolvesBeforeTheBodyIsRead(t *testing.T) { t.Run("a body over the cap", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.maxRequestBytes = 50 // below the body - req := rawIngestRequest(t, "clicks", "text/csv", + req := rawIngestRequest(t, "clicks", "application/xml", `{"page":"/`+strings.Repeat("a", 200)+`"}`) w := httptest.NewRecorder() h.Handle(w, withTenant(req)) @@ -2171,15 +2545,15 @@ func TestIngest_ContentTypeResolvesBeforeTheBodyIsRead(t *testing.T) { } // tsRegistry returns a registry whose events table carries the timestamp column -// shapes the #372 canonicalization path rewrites. +// shapes #372 is about: second and sub-second precision, in UTC. func tsRegistry(t testing.TB) *discovery.SchemaRegistry { return testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ { Name: "events", Columns: []discovery.Column{ {Name: "name", Type: "String"}, - {Name: "ts", Type: "DateTime('UTC')", HasDefault: true}, - {Name: "ts_ms", Type: "DateTime64(3, 'UTC')", HasDefault: true}, + {Name: "ts", Type: "DateTime('UTC')", HasDefault: true, DefaultExpression: "now()"}, + {Name: "ts_ms", Type: "DateTime64(3, 'UTC')", HasDefault: true, DefaultExpression: "now64(3)"}, }, }, }) @@ -2213,13 +2587,17 @@ func publishedRow(t *testing.T, payload []byte) map[string]any { return out } -// TestIngest_TimestampsCanonicalized is the #372 contract: whatever spelling a -// producer uses, the published payload — the one copy SSE subscribers, the -// ClickHouse insert, and the DLQ all consume — carries RFC 3339 UTC. +// TestIngest_TimestampsCanonicalized is the #372 contract, now satisfied by +// construction: whatever spelling a producer uses, the published payload — the +// one copy SSE subscribers, the ClickHouse insert and the DLQ all consume — +// carries the instant as ClickHouse's OWN writer renders it. That is +// "2026-06-21 04:00:00" in the column's zone, not RFC 3339 with a Z: the row is +// the server's rendering of the stored value, so the SSE frame and a +// /v1/query row cannot disagree. func TestIngest_TimestampsCanonicalized(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(tsRegistry(t)), pub) + h := newTestIngestHandler(t, tsRegistry(t), pub) req := ingestRequest(t, "events", map[string]any{ "name": "e", @@ -2231,24 +2609,24 @@ func TestIngest_TimestampsCanonicalized(t *testing.T) { require.Equal(t, http.StatusOK, w.Code) data := publishedData(t, pub) - assert.Equal(t, "2026-06-21T04:00:00Z", data["ts"]) - assert.Equal(t, "2026-06-21T04:00:00.5Z", data["ts_ms"]) + assert.Equal(t, "2026-06-21 04:00:00", data["ts"]) + // Sub-second digits are the column's precision, zeros and all — the server + // does not trim them the way the old canonicalizer did. + assert.Equal(t, "2026-06-21 04:00:00.500", data["ts_ms"]) assert.Equal(t, "e", data["name"], "non-timestamp columns untouched") } -// TestIngest_AutoInjectedLiteralTimestampCanonicalized pins the LiteralValue -// unwrap on the auto-inject path: a placeholder-free _eq check value is typed -// policy.LiteralValue for the comparison, but must enter the published data as -// a plain string — timestamp canonicalization switches on `case string`, so a -// leaked named type would silently skip the rewrite and publish the -// non-canonical spelling (the #372 fail-open that #381's row filter relies -// on). json.Marshal renders both identically, so only this assertion on the -// canonical form catches the leak. +// TestIngest_AutoInjectedLiteralTimestampCanonicalized: a value the policy +// auto-injects is not a shortcut past the parser. The check literal is written +// in a spelling ClickHouse does not store it in, so the published row proves +// the injected value went through the same parse every producer-supplied value +// does — an injected value published in its policy spelling would mean the +// server and the stream disagree about what the row holds. func TestIngest_AutoInjectedLiteralTimestampCanonicalized(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(tsRegistry(t)), pub) - staticTS := "2026-06-21 04:00:00" + h := newTestIngestHandler(t, tsRegistry(t), pub) + staticTS := "2026-06-21T04:00:00Z" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "events": { @@ -2259,8 +2637,8 @@ func TestIngest_AutoInjectedLiteralTimestampCanonicalized(t *testing.T) { }, }) - // ts omitted from the body — the static literal is auto-injected, then - // canonicalized like any producer-supplied spelling. + // ts omitted from the body — the static literal is injected as the column's + // DEFAULT, then parsed like any producer-supplied spelling. req := ingestRequest(t, "events", map[string]any{"name": "e"}) ctx := auth.WithRole(req.Context(), "user") ctx = auth.WithClaims(ctx, jwt.MapClaims{}) @@ -2270,31 +2648,38 @@ func TestIngest_AutoInjectedLiteralTimestampCanonicalized(t *testing.T) { h.Handle(w, withTenant(req)) require.Equal(t, http.StatusOK, w.Code) - assert.Equal(t, "2026-06-21T04:00:00Z", publishedData(t, pub)["ts"], - "auto-injected literal must be canonicalized, not published in its policy spelling") + assert.Equal(t, "2026-06-21 04:00:00", publishedData(t, pub)["ts"], + "auto-injected literal must be parsed, not published in its policy spelling") } -// TestIngest_TimestampGarbage_PassesThrough: fail-open — an unparseable value -// publishes verbatim; ClickHouse's own parser decides insertability (#372/#381). -func TestIngest_TimestampGarbage_PassesThrough(t *testing.T) { +// TestIngest_TimestampGarbage_Rejected: the old fail-open is gone. An +// unparseable timestamp used to publish verbatim and fail later at the worker's +// INSERT, where the caller could not see it; ClickHouse's own parser now +// answers at the edge, with its own code, and nothing is published. +func TestIngest_TimestampGarbage_Rejected(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(tsRegistry(t)), pub) + h := newTestIngestHandler(t, tsRegistry(t), pub) req := ingestRequest(t, "events", map[string]any{"name": "e", "ts": "banana"}) w := httptest.NewRecorder() h.Handle(w, withTenant(req)) - require.Equal(t, http.StatusOK, w.Code) - assert.Equal(t, "banana", publishedData(t, pub)["ts"], "unparseable value published verbatim") + require.Equal(t, http.StatusBadRequest, w.Code) + msg, code := errorAndCode(t, w) + assert.Contains(t, msg, "Cannot read DateTime") + assert.Equal(t, 41, code, "ClickHouse's own code rides with the message") + assert.Empty(t, pub.Messages) } -// TestIngest_Batch_MixedTimestampSpellings: parseable spellings canonicalize, -// the unparseable one passes through — no record fails on its timestamp. +// TestIngest_Batch_MixedTimestampSpellings: every parseable spelling lands on +// the same instant in the server's own rendering, and the unparseable one fails +// ALONE — one bad timestamp in a batch must not cost its siblings, which is +// what the compile profile's allow_errors_ratio buys. func TestIngest_Batch_MixedTimestampSpellings(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(tsRegistry(t)), pub) + h := newTestIngestHandler(t, tsRegistry(t), pub) req := ingestRequest(t, "events", []map[string]any{ {"name": "a", "ts": "2026-06-21T04:00:00Z"}, @@ -2305,25 +2690,21 @@ func TestIngest_Batch_MixedTimestampSpellings(t *testing.T) { h.Handle(w, withTenant(req)) require.Equal(t, http.StatusOK, w.Code) - var result struct { - Total int `json:"total"` - Succeeded int `json:"succeeded"` - Failed int `json:"failed"` - } - require.NoError(t, json.Unmarshal(w.Body.Bytes(), &result)) + result := decodeBatchResult(t, w) + require.Len(t, result.Results, 3) + assert.Equal(t, 41, result.Results[1].ExceptionCode, "the failing record carries ClickHouse's code") assert.Equal(t, 3, result.Total) - assert.Equal(t, 3, result.Succeeded) - assert.Equal(t, 0, result.Failed) - require.Len(t, pub.Messages, 3, "every record published") + assert.Equal(t, 2, result.Succeeded) + assert.Equal(t, 1, result.Failed) + require.Len(t, pub.Messages, 2, "only the records ClickHouse accepted") var spellings []string for _, msg := range pub.Messages { spellings = append(spellings, publishedRow(t, msg.Data)["ts"].(string)) } assert.Equal(t, []string{ - "2026-06-21T04:00:00Z", // already canonical - "banana", // unparseable — passed through verbatim - "2026-06-21T04:00:00Z", // Unix seconds — canonicalized + "2026-06-21 04:00:00", // RFC 3339 in + "2026-06-21 04:00:00", // Unix seconds in — same instant, same rendering }, spellings) } @@ -2346,7 +2727,7 @@ func TestIngest_Dedup_DisabledBySettings(t *testing.T) { pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() dedup.Err = errors.New("must not be called while disabled") - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{IDField: "event_id", RequireID: true} @@ -2368,7 +2749,7 @@ func TestIngest_Dedup_DisabledMidReload(t *testing.T) { pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() dedup.Err = dedupe.ErrDisabled - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: true} @@ -2388,22 +2769,21 @@ func TestIngest_Dedup_DisabledMidReload(t *testing.T) { // so the condition is true for every record or none; as a reject, a 10k batch // would report 10k independent permission failures for one mis-wired grant. // -// Second, the reachability claim at the CheckClauses call site. The column loop -// above it iterates the RECORD's columns, so it runs zero times for `{}` — and -// discovery.Validate accepts `{}` here because every column is nullable or -// defaulted. I previously asserted this path was unreachable, having tested only -// against a schema with a required column; it is not. -func TestPrepareRecord_UnresolvedInsertSideAborts(t *testing.T) { +// Second, the reachability claim at the CheckClauses call site: nothing +// rejects an empty record before it, because ClickHouse reads `{}` as every +// column taking its default, so under the type layer it is reachable for EVERY +// table. (It was once asserted unreachable, having been tested only against a +// schema with a required column.) +func TestInsertShape_UnresolvedInsertSideAborts(t *testing.T) { t.Parallel() schema := &discovery.TableSchema{ Name: "loose", Columns: []discovery.Column{ {Name: "a", Type: "Nullable(String)", IsNullable: true}, - {Name: "b", Type: "String", HasDefault: true}, + {Name: "b", Type: "String", HasDefault: true, DefaultExpression: "''"}, }, } - reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{schema}) - h := NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}) + h := NewIngestHandler(fixedRegistry(testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{schema})), &testutil.MockPublisher{}) // A grant resolved for SELECT, reaching the insert path. selectResolved := policy.Evaluate(&policy.Policy{ @@ -2413,14 +2793,10 @@ func TestPrepareRecord_UnresolvedInsertSideAborts(t *testing.T) { }, "viewer", "loose", "select", nil) require.True(t, selectResolved.Allowed) - // An empty record really does clear validation and the column loop. - require.NoError(t, discovery.Validate(schema, map[string]any{}), - "all-nullable/defaulted columns accept an empty record — this is what makes the read reachable") - - rec, abort := h.prepareRecord( - context.Background(), testStore, "loose", "", schema, selectResolved, "viewer", map[string]any{}, time.Now(), nil) + _, preds, cols, abort := h.insertShape(context.Background(), "loose", "viewer", schema, selectResolved, nil) - assert.Nil(t, rec.reject, "a request-scoped condition must not be reported per record") + assert.Nil(t, preds, "an unresolved insert side must produce no check predicates") + assert.Nil(t, cols) require.NotNil(t, abort, "an unresolved insert side must abort the request") assert.Equal(t, http.StatusForbidden, abort.Status) assert.Empty(t, abort.RetryAfter, "not a transient condition — retrying cannot help") @@ -2460,7 +2836,7 @@ func TestIngest_ContentTypeEchoIsBounded(t *testing.T) { t.Run(name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(build(t))) @@ -2488,7 +2864,7 @@ func TestIngest_ContentTypeEchoIsBounded(t *testing.T) { req.Header.Add("Content-Type", fmt.Sprintf("application/%04d", i)+strings.Repeat("\xff", 112)) } w := httptest.NewRecorder() - NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}).Handle(w, withTenant(req)) + newTestIngestHandler(t, testRegistry(t), &testutil.MockPublisher{}).Handle(w, withTenant(req)) require.Equal(t, http.StatusUnsupportedMediaType, w.Code) return w.Body.Len() } @@ -2512,7 +2888,7 @@ func TestIngest_ContentTypeEchoIsBounded(t *testing.T) { func TestIngest_ConflictMessageNamesTheDisagreement(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "", "{\"page\":\"/a\"}\n{\"page\":\"/b\"}") for range 4 { req.Header.Add("Content-Type", "application/json") @@ -2542,7 +2918,7 @@ func TestIngest_ConflictMessageNamesTheDisagreement(t *testing.T) { func TestIngest_ConflictMessageNamesADifferentSpelling(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) req := rawIngestRequest(t, "clicks", "", `{"page":"/a"}`) for _, ct := range []string{ "application/json", @@ -2572,7 +2948,7 @@ func TestIngest_ConflictMessageNamesADifferentSpelling(t *testing.T) { // declaration buried. Nothing covered log CONTENT, because the tests discarded it. func TestIngest_ConflictLogNamesTheDisagreement(t *testing.T) { buf := logtest.Capture(t, slog.LevelInfo) - h := NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}) + h := newTestIngestHandler(t, testRegistry(t), &testutil.MockPublisher{}) req := rawIngestRequest(t, "clicks", "", `{"page":"/a"}`) for _, ct := range []string{ @@ -2600,7 +2976,7 @@ func TestIngest_ConflictLogNamesTheDisagreement(t *testing.T) { func TestIngest_CheckColumnNotInSchema_Rejected(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "tenant_id", "acme")) w := httptest.NewRecorder() @@ -2633,7 +3009,7 @@ func TestIngest_CheckOnComputedColumn_Rejected(t *testing.T) { }}}, }} pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(computedRegistry(t)), pub) + h := newTestIngestHandler(t, computedRegistry(t), pub) h.PolicySource = staticPolicy(p) w := httptest.NewRecorder() @@ -2651,12 +3027,12 @@ func TestIngest_CheckOnComputedColumn_Rejected(t *testing.T) { // TestIngest_CheckColumnInSchema_StillInjects is the other half: the guard must // not disturb the case it sits next to. A record omitting a checked column that -// DOES exist still gets the value injected, canonicalized, and carried in the -// encoded row at that column's position. +// DOES exist still gets the value injected and carried in the exported row at +// that column's position. func TestIngest_CheckColumnInSchema_StillInjects(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "org_id", "org-42")) w := httptest.NewRecorder() @@ -2681,7 +3057,7 @@ func TestIngest_CheckColumnInSchema_StillInjects(t *testing.T) { func TestIngest_CheckGuardLogsOncePerRequest(t *testing.T) { buf := logtest.Capture(t, slog.LevelInfo) pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "tenant_id", "acme")) const n = 25 @@ -2706,14 +3082,17 @@ func TestIngest_CheckGuardLogsOncePerRequest(t *testing.T) { // // Every record fails here, and that is not an artifact of the fixture: the // condition is a property of (table, role, policy), identical for every record -// in the request, so there is no sibling this guard could spare. A record -// SUPPLYING the missing column is rejected too, one step earlier, by schema -// validation — with its own message, which this pins so the two stay -// distinguishable. +// in the request, so there is no sibling this guard could spare. +// +// CONTRACT CHANGE: a record SUPPLYING the missing column used to get the +// guard's message too, because the gateway's key walk ran first. ClickHouse's +// parser now answers first, so that record reports its own code 117 — the same +// ordering change as everywhere else on this path. Both are still +// failures and nothing publishes. func TestIngest_CheckColumnNotInSchema_BatchRejectsPerRecord(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "tenant_id", "acme")) req := rawIngestRequest(t, "clicks", "application/json", @@ -2729,9 +3108,13 @@ func TestIngest_CheckColumnNotInSchema_BatchRejectsPerRecord(t *testing.T) { assert.Equal(t, 3, resp.Failed) assert.Zero(t, resp.Succeeded) require.Len(t, resp.Results, 3) - assert.Contains(t, resp.Results[0].Error, "which table", "omitted ⇒ the new guard") - assert.Contains(t, resp.Results[1].Error, "unknown column", "supplied ⇒ schema validation, one step earlier") - assert.Contains(t, resp.Results[2].Error, "which table") + for _, i := range []int{0, 2} { + assert.Contains(t, resp.Results[i].Error, "which table", "record %d", i+1) + assert.Zero(t, resp.Results[i].ExceptionCode, "a gateway rejection never carries a ClickHouse code") + } + assert.Contains(t, resp.Results[1].Error, "tenant_id") + assert.Equal(t, 117, resp.Results[1].ExceptionCode, + "the record that SUPPLIES the column is refused by ClickHouse first") assert.Empty(t, pub.Messages) } @@ -2753,10 +3136,10 @@ func computedRegistry(t testing.TB) *discovery.SchemaRegistry { Name: "clicks", Columns: []discovery.Column{ {Name: "page", Type: "String"}, - {Name: "digest", Type: "String", DefaultKind: "MATERIALIZED", HasDefault: true}, - {Name: "country", Type: "String", HasDefault: true}, - {Name: "doubled", Type: "UInt64", DefaultKind: "ALIAS", HasDefault: true}, - {Name: "raw", Type: "String", DefaultKind: "EPHEMERAL", HasDefault: true}, + {Name: "digest", Type: "String", DefaultKind: "MATERIALIZED", HasDefault: true, DefaultExpression: "upper(page)"}, + {Name: "country", Type: "String", HasDefault: true, DefaultExpression: "'US'"}, + {Name: "doubled", Type: "UInt64", DefaultKind: "ALIAS", HasDefault: true, DefaultExpression: "length(page) * 2"}, + {Name: "raw", Type: "String", DefaultKind: "EPHEMERAL", HasDefault: true, DefaultExpression: "''"}, }, }, }) @@ -2772,7 +3155,7 @@ func computedRegistry(t testing.TB) *discovery.SchemaRegistry { func TestIngest_CheckOnEphemeralColumn_Rejected(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(fixedRegistry(computedRegistry(t)), pub) + h := newTestIngestHandler(t, computedRegistry(t), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "raw", "anything")) w := httptest.NewRecorder() @@ -2788,7 +3171,7 @@ func TestIngest_CheckOnEphemeralColumn_Rejected(t *testing.T) { // event_id and dedup as the store. func dedupHandler(t *testing.T, pub *testutil.MockPublisher, dedup dedupe.Deduplicator, requireID bool) *IngestHandler { t.Helper() - h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h := newTestIngestHandler(t, testRegistry(t), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id", RequireID: requireID} diff --git a/internal/api/ingest_unavailable_test.go b/internal/api/ingest_unavailable_test.go new file mode 100644 index 00000000..4c6189f4 --- /dev/null +++ b/internal/api/ingest_unavailable_test.go @@ -0,0 +1,202 @@ +package api + +import ( + "errors" + "log/slog" + "net/http" + "net/http/httptest" + "testing" + "testing/iotest" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/auth" + "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/policy" + "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil" + "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" + "github.com/Wave-RF/WaveHouse/internal/typelayer" +) + +// The type layer judges every record, so there is no Go-side comparison left +// to swap out. What these tests pin is the type layer being absent or unable to +// answer for a tenant — the fail-closed property — and the ordering it imposes. + +// assertTypesUnavailable pins the 503 a tenant the type layer cannot judge +// answers with: a generic message, the schema refresh's Retry-After, nothing +// published. +func assertTypesUnavailable(t *testing.T, w *httptest.ResponseRecorder, pub *testutil.MockPublisher) { + t.Helper() + require.Equal(t, http.StatusServiceUnavailable, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, retryAfterSchema, w.Header().Get("Retry-After"), "an outage is retryable; the body is unchanged") + assert.Equal(t, "ingest validation is unavailable", jsonErrorMessage(t, w)) + testutil.AssertJSONErrorResponse(t, w) + assert.Empty(t, pub.Messages) +} + +// TestIngest_UnwiredTypeLayer_Refuses: a handler with no type layer must refuse +// the request, not wave it through. Nothing else on the ingest path looks at a +// value, so nil Types reading as "accept anything" would turn a wiring mistake +// into an open door — the same fail-closed direction the row evaluator takes. +func TestIngest_UnwiredTypeLayer_Refuses(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + require.Nil(t, h.Types) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home"}))) + + assertTypesUnavailable(t, w, pub) +} + +// TestIngest_UndiscoveredTable_Unavailable: a table the registry knows but the +// type layer has no compiled schema for is a 503, never a 400 — the request is +// fine; the server cannot judge it. The cause is the operator's: it is logged, +// and the body stays generic. +func TestIngest_UndiscoveredTable_Unavailable(t *testing.T) { + buf := logtest.Capture(t, slog.LevelError) + pub := &testutil.MockPublisher{} + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) + h.Types = typelayer.TestEngine(t) // bound to NO tables + + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home"}))) + + assertTypesUnavailable(t, w, pub) + assert.Contains(t, buf.String(), "not discovered", "the log carries the cause") +} + +// TestIngest_UnboundTenant_RefusedBeforeTheBodyIsRead: a tenant the type layer +// has not bound — its discovery has not refreshed, its line has no artifact, its +// zone differs from the one this process opened the line in — is refused on its +// own, with the schema refresh's Retry-After, and decided before the body is +// read, like a schema not discovered yet. Another tenant on the same engine +// keeps ingesting, and the body never names the tenant or the cause. +func TestIngest_UnboundTenant_RefusedBeforeTheBodyIsRead(t *testing.T) { + t.Parallel() + tenants := nestedTenants(t, map[string]string{"acme": fullConfig(100), "globex": fullConfig(100)}) + reg := testRegistry(t) + pub := &testutil.MockPublisher{} + h := NewIngestHandler(fixedRegistry(reg), pub) + h.Types = typelayer.TestEngine(t) // tenant.Default only, and no tables + h.Types.Bind("globex", typelayer.TestServerVersion, "UTC", reg.List()) + + acme, ok := tenants.For("acme") + require.True(t, ok) + req := httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/v1/ingest?table=clicks", + iotest.ErrReader(errors.New("the body must not be read"))) + req.Header.Set("Content-Type", "application/json") + w := httptest.NewRecorder() + h.Handle(w, req.WithContext(WithStore(req.Context(), acme))) + assertTypesUnavailable(t, w, pub) + assert.NotContains(t, w.Body.String(), "acme", "the body does not name the tenant") + + globex, ok := tenants.For("globex") + require.True(t, ok) + req = ingestRequest(t, "clicks", map[string]any{"page": "/home"}) + w = httptest.NewRecorder() + h.Handle(w, req.WithContext(WithStore(req.Context(), globex))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + require.Len(t, pub.Published(), 1) + assert.Equal(t, tenant.ID("globex"), pub.Published()[0].Topic.Tenant) +} + +// TestIngest_ParseErrorsPrecedeCheckErrors is a DOCUMENTED contract change. +// Check clauses are a filter over rows ClickHouse has already accepted, so a +// record that fails both reports the PARSE error, not the check. No enforcement +// is lost — nothing is published either way — but a client can no longer infer +// from a 403 that the rest of its payload was well formed. +func TestIngest_ParseErrorsPrecedeCheckErrors(t *testing.T) { + t.Parallel() + required := "org-allowed" + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + h.PolicySource = staticPolicy(&policy.Policy{Tables: map[string]policy.TablePolicy{ + "clicks": {"viewer": {Insert: &policy.InsertPermissions{ + Check: map[string]policy.Filter{"org_id": {Eq: &required}}, + }}}, + }}) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{ + "page": "/a", + "org_id": "wrong", // fails the check clause + "count": "not-a-number", // and ClickHouse refuses it first + }))) + + require.Equal(t, http.StatusBadRequest, w.Code, "body=%s", w.Body.String()) + _, code := errorAndCode(t, w) + assert.Equal(t, 27, code, "ClickHouse's own refusal, with its own code") + assert.Empty(t, pub.Messages) + + // The same record, parseable: NOW the check clause is what refuses it. + w = httptest.NewRecorder() + h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{ + "page": "/a", + "org_id": "wrong", + "count": 1, + }))) + assert.Equal(t, http.StatusForbidden, w.Code, "body=%s", w.Body.String()) + assert.Contains(t, jsonErrorMessage(t, w), `check failed for column "org_id"`) + _, code = errorAndCode(t, w) + assert.Zero(t, code, "a gateway rejection never carries a ClickHouse code") + assert.Empty(t, pub.Messages) +} + +// TestIngest_DedupeMarksOnlyPublishedRecords: dedupe runs AFTER validation, so a +// record ClickHouse refuses never claims its id — the caller can fix the record +// and resend it under the same id. Claiming before validation would swallow the +// corrected retry as a duplicate (or as in flight, for a lease). +func TestIngest_DedupeMarksOnlyPublishedRecords(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + dedup := testutil.NewMockDeduplicator() + h := newTestIngestHandler(t, testRegistry(t), pub) + h.Dedup = staticDedup(dedup) + h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { + return settings.Dedupe{Enabled: true, IDField: "event_id"} + } + key := dedupe.Key{Table: "clicks", ID: "evt-1"} + + // A record ClickHouse refuses (count is not a number), carrying an id. + w := httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/a", "event_id": "evt-1", "count": "nope"}))) + require.Equal(t, http.StatusBadRequest, w.Code, "body=%s", w.Body.String()) + assert.Empty(t, pub.Messages) + assert.Zero(t, dedup.Reserves, "a refused record reserves nothing") + assert.False(t, dedup.Pending(key)) + assert.False(t, dedup.Committed(key)) + + // The same id, corrected: it must publish, not report a duplicate. + w = httptest.NewRecorder() + h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/a", "event_id": "evt-1", "count": 1}))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Contains(t, w.Body.String(), `"ok":true`) + assert.Len(t, pub.Messages, 1) + + // And only NOW is the id spent. + assert.True(t, dedup.Committed(key), "the published record's id is committed") + + // In a batch, the refused sibling claims nothing either. + w = httptest.NewRecorder() + h.Handle(w, withTenant(ndjsonRequest(t, "clicks", + jsonLine(t, map[string]any{"page": "/b", "event_id": "evt-2", "count": "nope"}), + jsonLine(t, map[string]any{"page": "/c", "event_id": "evt-3"}), + ))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.False(t, dedup.Pending(dedupe.Key{Table: "clicks", ID: "evt-2"})) + assert.False(t, dedup.Committed(dedupe.Key{Table: "clicks", ID: "evt-2"})) + assert.True(t, dedup.Committed(dedupe.Key{Table: "clicks", ID: "evt-3"})) +} + +// viewerIngestRequest is ingestRequest with the "viewer" role in context, for +// the policy-gated tests. +func viewerIngestRequest(t *testing.T, table string, body map[string]any) *http.Request { + t.Helper() + req := ingestRequest(t, table, body) + return req.WithContext(auth.WithRole(req.Context(), "viewer")) +} diff --git a/internal/api/ingest_window_test.go b/internal/api/ingest_window_test.go index 1fd13fb6..5158bcd8 100644 --- a/internal/api/ingest_window_test.go +++ b/internal/api/ingest_window_test.go @@ -237,7 +237,7 @@ func TestIngest_Windows_OutcomesStayInOrder(t *testing.T) { assert.Equal(t, []recordResult{ {Index: 1, Ok: true}, {Index: 2, Duplicate: true}, - {Index: 3, Error: resp.Results[2].Error}, + {Index: 3, Error: resp.Results[2].Error, ExceptionCode: 117}, {Index: 4, Ok: true}, {Index: 5, Duplicate: true}, {Index: 6, Ok: true}, @@ -394,7 +394,7 @@ func pebbleBatchHandler(tb testing.TB, window int) (*IngestHandler, *countingDed require.NoError(tb, store.Apply(true)) tb.Cleanup(func() { _ = store.Close() }) counted := &countingDedup{Deduplicator: store} - h := NewIngestHandler(fixedRegistry(testRegistry(tb)), &testutil.MockPublisher{}) + h := newTestIngestHandler(tb, testRegistry(tb), &testutil.MockPublisher{}) h.Dedup = staticDedup(counted) h.DedupeSettings = func(*settings.Store, string) settings.Dedupe { return settings.Dedupe{Enabled: true, IDField: "event_id"} From 1ceb21d4d4860387b0ea908475b4c4a3c761d57f Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:07:50 -0400 Subject: [PATCH 12/70] test(e2e)!: pin ClickHouse's own ingest verdicts and exception_code The SDK's InsertRecordResult gains exception_code, ClickHouse's numeric error code on a per-record parser refusal; the string code class is unchanged. The ingest, NDJSON, DLQ and batching suites now assert the type layer's behaviour: CSV/TSV with the header parameter, a compact JSON array that keeps its good records, a denied column as 117, a header refusal as clickhouse.rejected with exception_code 117, an _in check judged on the table default, and an unparseable row refused at ingest so the DLQ never sees it. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- clients/ts/src/types.ts | 24 ++- tests/e2e/sdk/batching.test.ts | 6 + tests/e2e/sdk/dlq.test.ts | 54 ++--- tests/e2e/sdk/ingest.test.ts | 346 ++++++++++++++++++++++++++++++++- tests/e2e/sdk/ndjson.test.ts | 66 ++++++- 5 files changed, 459 insertions(+), 37 deletions(-) diff --git a/clients/ts/src/types.ts b/clients/ts/src/types.ts index 158d5c44..bfd88ba9 100644 --- a/clients/ts/src/types.ts +++ b/clients/ts/src/types.ts @@ -292,9 +292,9 @@ export type Schemas = Record; // --- Insert result --- /** - * A per-record outcome from a batch (array / NDJSON) insert. Mirrors the - * single-object response shape plus the record's position. Exactly one of - * `ok` / `duplicate` / `error` is set. + * A per-record outcome from a batch (array / NDJSON / CSV / TSV) insert. + * Mirrors the single-object response shape plus the record's position. Exactly + * one of `ok` / `duplicate` / `error` is set. */ export interface InsertRecordResult { /** 1-based index of the record within the submitted batch. */ @@ -305,6 +305,22 @@ export interface InsertRecordResult { duplicate?: boolean; /** Set (with `ok`/`duplicate` absent) when the record was rejected. */ error?: string; + /** + * ClickHouse's own numeric error code, present only when the server's parser + * is what refused the record — 117 unknown field, 27 unparseable value, 6 out + * of range. Absent for a gateway rejection (a failed policy check, a missing + * dedupe id), so `exception_code !== undefined` means "ClickHouse answered". + * The same name carries it on a whole-request error body, beside the string + * `code` class (reachable as `error.details`). + * + * 117 also covers **a column the caller's role may not write**. Column policy + * is enforced by compiling the role's own schema without the denied columns, + * so naming one is an unknown field to the parser rather than a separate + * gateway refusal: a `400` with this code, where it used to be a + * `403 column "x" not allowed for insert`. The message is ClickHouse's own and + * does not reveal whether the column exists. + */ + exception_code?: number; } export interface InsertResult { @@ -320,7 +336,7 @@ export interface InsertResult { failed?: number; /** Batch insert: records skipped by dedup. */ duplicates?: number; - /** Batch insert: per-record outcomes, each `{index, ok|duplicate|error}` (may be truncated for very large batches; the counts stay authoritative). */ + /** Batch insert: per-record outcomes, each `{index, ok|duplicate|error, exception_code?}` (may be truncated for very large batches; the counts stay authoritative). */ results?: InsertRecordResult[]; } diff --git a/tests/e2e/sdk/batching.test.ts b/tests/e2e/sdk/batching.test.ts index bc2e8fc0..8596f3d6 100644 --- a/tests/e2e/sdk/batching.test.ts +++ b/tests/e2e/sdk/batching.test.ts @@ -2,6 +2,12 @@ import { describe, expect, it } from "vitest"; import { chQuery, dataClient, testId, waitForCondition } from "./helpers.js"; import { suiteTables } from "./tables.js"; +// These trigger the INGEST WORKER's flush (500 rows or a 5s linger). A body is +// one type-layer call whatever its size, and the handler's dedupe windows (256 +// records) only pace reserve/publish/commit, so the 500 here is the worker's +// number and nothing in the handler shares it — a batch of any size comes back +// as one contiguous, 1-based result set +// (TestIngest_LargeBatch_IndicesStayContiguous pins the indexing cheaply). describe("Ingest Batching Triggers", () => { const wh = dataClient(); const T = suiteTables("batching"); diff --git a/tests/e2e/sdk/dlq.test.ts b/tests/e2e/sdk/dlq.test.ts index 1421b894..b9ecf80e 100644 --- a/tests/e2e/sdk/dlq.test.ts +++ b/tests/e2e/sdk/dlq.test.ts @@ -7,25 +7,31 @@ describe("Dead Letter Queue (DLQ) & Failures", () => { const admin = adminClient(); const T = suiteTables("dlq"); - it("routes only the failed row to DLQ while valid rows are inserted", async () => { + // This case used to assert the opposite: a row that "bypasses API validation + // but fails database insertion" landing in the DLQ. That gap is what the type + // layer closed — ingest now asks ClickHouse's own parser before publishing, + // so an unparseable value is refused at the gateway with the server's real + // code and never enters the queue. The DLQ still exists for failures that + // only surface at INSERT time (covered by tests/integration/dlq_test.go and + // internal/ingest/worker_test.go); what this pins is that bad data no longer + // gets that far. + it("refuses the unparseable row at ingest, so the DLQ never sees it", async () => { const runId = testId(); - // Get baseline DLQ stats before we pollute them. DLQ stats are keyed by - // table name, and this suite owns T.clicks exclusively, so the count is - // isolated from every other test file. + // Baseline before we touch it. DLQ stats are keyed by table name and this + // suite owns T.clicks exclusively, so the count is isolated from every + // other test file. const initialDlq = await admin.dlq.list(); const initialClicksDlq = (initialDlq.data?.tables as any)?.[T.clicks] || 0; - // We are going to send 9 perfectly valid rows, and 1 critically malformed row. + // Nine valid rows and one whose duration_ms no ClickHouse parser can read. const rows = Array.from({ length: 10 }).map((_, i) => { if (i === 9) { return { event_id: `${runId}-bad`, - page: "/bag-page", + page: "/bad-page", session_id: `session-${runId}`, user_id: `user-${runId}`, - // Go accepts strings for numerics, but ClickHouse cannot parse this into an Int/Float. - // This successfully bypasses API validation but fails database insertion. duration_ms: "definitely-not-a-number", }; } @@ -38,15 +44,18 @@ describe("Dead Letter Queue (DLQ) & Failures", () => { }); const res = await wh.from(T.clicks).insert(rows as any); - expect(res.error).toBeNull(); // API accepts it (schema validation is loose by design) + // A per-record rejection is not a request failure: the batch is a 200 whose + // body carries one verdict per record. + expect(res.error).toBeNull(); + expect(res.data?.succeeded).toBe(9); + expect(res.data?.failed).toBe(1); + const refused = res.data?.results?.find((r) => r.error); + expect(refused?.index).toBe(10); + // An `exception_code` means ClickHouse answered — a gateway rejection + // carries none. + expect(refused?.exception_code).toBeGreaterThan(0); - // 9 good rows must land. Budget is 10s (the suite norm), not the 6s a plain - // timer-flush uses: the bad row forces the worker into 1-by-1 isolation — - // ~11 sequential CH round-trips after the 5s maxWait timer — which is far - // more post-timer work than a single-insert flush, so 6s sat right at the - // edge under CI load. The structural fix is a lower e2e maxWait (deferred - // config PR), which drops the 5s-timer dependency; this budget can shrink - // back once that lands. + // The nine good rows still land: one bad record never blocks the batch. await waitForCondition(async (signal) => { const chRows = await chQuery( `SELECT count() as cnt FROM default.${T.clicks} WHERE user_id = 'user-${runId}'`, @@ -55,17 +64,10 @@ describe("Dead Letter Queue (DLQ) & Failures", () => { return Number((chRows[0] as any).cnt) === 9; }, 10_000); - // Verify exactly 1 message was added to the DLQ for this suite's clicks table - await waitForCondition(async () => { - const dlqRes = await admin.dlq.list(); - const currentClicksDlq = (dlqRes.data?.tables as any)?.[T.clicks] || 0; - return currentClicksDlq === initialClicksDlq + 1; - }, 5_000); - + // The refused row was never published, so nothing can have reached the DLQ. + // The wait above already proves the worker drained this batch. const finalDlq = await admin.dlq.list(); const finalClicksDlq = (finalDlq.data?.tables as any)?.[T.clicks] || 0; - - // Only 1 was rejected and routed to the DLQ - expect(finalClicksDlq).toBe(initialClicksDlq + 1); + expect(finalClicksDlq).toBe(initialClicksDlq); }, 20_000); }); diff --git a/tests/e2e/sdk/ingest.test.ts b/tests/e2e/sdk/ingest.test.ts index d540147a..9eb0a0a6 100644 --- a/tests/e2e/sdk/ingest.test.ts +++ b/tests/e2e/sdk/ingest.test.ts @@ -72,6 +72,9 @@ describe("Ingest", () => { expect(inCH).toHaveLength(3); }); + // The unknown field is refused by ClickHouse's own parser (code 117): the + // compile profile pins input_format_skip_unknown_fields=0 precisely so this + // stays a verdict the caller hears rather than silent data loss. it("rejects unknown fields with a validation error", async () => { const result = await wh.from(T.clicks).insert({ event_id: testId(), @@ -83,6 +86,7 @@ describe("Ingest", () => { expect(result.error).not.toBeNull(); expect(result.error!.status).toBe(400); + expect((result.error!.details as { exception_code?: number }).exception_code).toBe(117); }); it("rejects ingest to a non-existent table", async () => { @@ -265,14 +269,50 @@ describe("Ingest", () => { await chQuery(`DROP TABLE IF EXISTS \`${maliciousName}\``); }); - it("rejects ingest containing the reserved received_timestamp field", async () => { + it("accepts an explicit received_timestamp and stores the caller's instant", async () => { + // This case used to assert a 400 and pass for the wrong reason: the payload + // omitted three columns that are neither nullable nor defaulted, and + // WaveHouse's own validator called that "missing required column". + // ClickHouse does not — an omitted JSONEachRow field takes the type's + // default — and the gateway now gives the server's answer. Nothing is + // reserved about received_timestamp on the way in: it is an ordinary column + // with a DEFAULT, and a caller that supplies one keeps it. + const id = testId(); + const result = await wh.from(T.clicks).insert({ + event_id: id, + received_timestamp: "2026-01-01T00:00:00.000", + } as any); + expect(result.error).toBeNull(); + + await waitForCondition(async () => { + const rows = await chQuery( + `SELECT page, toString(received_timestamp) AS ts FROM default.${T.clicks} WHERE event_id = '${id}'`, + ); + if (rows.length !== 1) return false; + expect(rows[0].ts).toBe("2026-01-01 00:00:00.000"); + expect(rows[0].page).toBe(""); + return true; + }, 15_000); + }); + + it("rejects a value ClickHouse cannot parse, with ClickHouse's own code", async () => { + // The replacement for the "reserved field" case above: a real per-record + // refusal, carrying the server's error code rather than a gateway guess. const result = await wh.from(T.clicks).insert({ event_id: testId(), - received_timestamp: "2026-01-01T00:00:00Z", + page: "/bad-duration", + user_id: "u1", + session_id: "s1", + duration_ms: "not-a-number", } as any); expect(result.error).not.toBeNull(); expect(result.error!.status).toBe(400); + // A per-record refusal carries ClickHouse's number as exception_code; the + // string `code` stays the failure class, which a per-record refusal has + // none of, so the SDK falls back to the status. + expect((result.error!.details as { exception_code?: number }).exception_code).toBe(27); + expect(result.error!.code).toBe("HTTP_400"); }); it("rejects invalid JSON payloads", async () => { @@ -288,17 +328,24 @@ describe("Ingest", () => { body: "{ bad json", }); expect(res.status).toBe(400); + // The refusal is ClickHouse's own parser now, so the body carries its + // exception_code rather than a flat gateway "invalid json". + const body = (await res.json()) as { error?: string; exception_code?: number }; + expect(typeof body.exception_code).toBe("number"); }); it("rejects an ingest that declares no readable Content-Type", async () => { - for (const headers of [ + const declarations: Record[] = [ // `fetch` supplies text/plain for a string body when none is set. { Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}` }, { Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}`, - "Content-Type": "text/csv", + // text/csv USED to sit here; it is an accepted format now, so the + // refusal case needs a type ingest genuinely does not read. + "Content-Type": "application/xml", }, - ]) { + ]; + for (const headers of declarations) { const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { method: "POST", headers, @@ -312,7 +359,7 @@ describe("Ingest", () => { // `application/jsonlines`. const body = (await res.json()) as { error?: string }; expect(body.error).toContain( - "application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines", + "application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent", ); } }); @@ -373,6 +420,293 @@ describe("Ingest", () => { await setPolicy(currentPolicy); }); + // CONTRACT CHANGE: a column the caller's role may not write is no longer a + // gateway 403 `column "x" not allowed for insert`. Column policy is enforced + // by compiling the role's own schema WITHOUT the denied columns, so naming one + // is ClickHouse's own per-record UNKNOWN_FIELD — a 400 with + // `exception_code: 117`, whose message does not say whether the column exists. + it("refuses a denied insert column with ClickHouse's code 117", async () => { + const currentPolicy = readPolicyFile(); + await setPolicy({ + tables: { + ...currentPolicy.tables, + [T.clicks]: { + ...(currentPolicy.tables[T.clicks] || {}), + viewer: { + ...(currentPolicy.tables[T.clicks]?.viewer || {}), + insert: { allow_columns: ["*"], deny_columns: ["country"] }, + }, + }, + }, + }); + + try { + const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { + Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + event_id: testId(), + page: "/denied-column", + user_id: "u1", + session_id: "s1", + country: "GB", + }), + }); + expect(res.status).toBe(400); + const body = (await res.json()) as { error?: string; exception_code?: number }; + expect(body.exception_code).toBe(117); + expect(body.error).toContain("country"); + + // The same role writing only permitted columns still succeeds, and the + // denied column takes the server's own default. + const okId = testId(); + const okRes = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { + Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + event_id: okId, + page: "/denied-column-ok", + user_id: "u1", + session_id: "s1", + }), + }); + expect(okRes.status).toBe(200); + await waitForCondition(async (signal) => { + const r = await chQuery( + `SELECT country FROM default.${T.clicks} WHERE event_id = '${okId}'`, + signal, + ); + return r.length === 1 && r[0].country === "US"; + }, 10_000); + } finally { + await setPolicy(currentPolicy); + } + }); + + // CSV and TSV are new accepted formats. They are positional + // in the table's declaration order — every wire column, in that order, with + // an empty field meaning "take the DEFAULT". An end-to-end assertion is the + // only one that catches a column-order bug: a mis-ordered body still answers + // 200. + it("ingests a complete positional CSV row end to end", async () => { + const id = testId(); + // Every wire column, in declaration order. An empty field takes the + // column's DEFAULT — which is how received_timestamp gets now64(3). + const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { + Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}`, + "Content-Type": "text/csv", + }, + body: `"${id}","/csv-full","u-csv","s-csv","","GB",7,\n`, + }); + expect(res.status).toBe(200); + const body = (await res.json()) as { succeeded: number; results?: unknown[] }; + expect(body.succeeded).toBe(1); + + await waitForCondition(async (signal) => { + const r = await chQuery( + `SELECT page, country, duration_ms FROM default.${T.clicks} WHERE event_id = '${id}'`, + signal, + ); + if (r.length !== 1) return false; + expect(r[0].page).toBe("/csv-full"); + expect(r[0].country).toBe("GB"); + expect(Number(r[0].duration_ms)).toBe(7); + return true; + }, 10_000); + }); + + // A bare text/csv is ClickHouse's default CSV: a first line that spells the + // column names is auto-detected as a header, so only the data row counts. + it("auto-detects a header line in bare CSV", async () => { + const id = testId(); + const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { + Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}`, + "Content-Type": "text/csv", + }, + body: + "event_id,page,user_id,session_id,referrer,country,duration_ms,received_timestamp\n" + + `"${id}","/csv-detect","u-d","s-d","","GB",7,\n`, + }); + expect(res.status).toBe(200); + const body = (await res.json()) as { total: number; succeeded: number }; + expect(body.total).toBe(1); + expect(body.succeeded).toBe(1); + }); + + it("ingests a complete positional TSV row, and a short row is code 27", async () => { + const id = testId(); + // TSV spells "take the DEFAULT" as ClickHouse's own \\N (null, which + // input_format_null_as_default turns into the column default). An EMPTY TSV + // field is the empty string, which a DateTime64 cannot read — unlike CSV, + // where an empty field IS the default. + const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { + Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}`, + "Content-Type": "text/tab-separated-values", + }, + // Row 1 is complete; row 2 is short, which is a PER-RECORD refusal with + // ClickHouse's code 27 — not a whole-request failure. + body: `${id}\t/tsv\tu-tsv\ts-tsv\t\tGB\t7\t\\N\nshort\t/tsv\n`, + }); + expect(res.status).toBe(200); + const body = (await res.json()) as { + total: number; + succeeded: number; + failed: number; + results: Array<{ index: number; exception_code?: number }>; + }; + expect(body.total).toBe(2); + expect(body.succeeded).toBe(1); + expect(body.failed).toBe(1); + expect(body.results[1].exception_code).toBe(27); + + await waitForCondition(async (signal) => { + const r = await chQuery( + `SELECT page, country FROM default.${T.clicks} WHERE event_id = '${id}'`, + signal, + ); + return r.length === 1 && r[0].page === "/tsv" && r[0].country === "GB"; + }, 10_000); + }); + + // `header=present` (RFC 4180's parameter, mirrored for TSV) selects the + // header formats: the first line names the columns in any order, it is not + // a record, and a column it omits takes its DEFAULT. + it("ingests CSV whose header names the columns in any order", async () => { + const id = testId(); + const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { + Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}`, + "Content-Type": "text/csv; header=present", + }, + body: `page,event_id,user_id,session_id\n/csv-names,${id},u-n,s-n\n`, + }); + expect(res.status).toBe(200); + const body = (await res.json()) as { total: number; succeeded: number }; + expect(body.total).toBe(1); + expect(body.succeeded).toBe(1); + + await waitForCondition(async (signal) => { + const r = await chQuery( + `SELECT page, user_id, country FROM default.${T.clicks} WHERE event_id = '${id}'`, + signal, + ); + return ( + r.length === 1 && + r[0].page === "/csv-names" && + r[0].user_id === "u-n" && + r[0].country === "US" + ); + }, 10_000); + }); + + it("ingests TSV with a header, and refuses an unknown header column with code 117", async () => { + const id = testId(); + const auth = { Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer" })}` }; + const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { ...auth, "Content-Type": "text/tab-separated-values; header=present" }, + body: `session_id\tevent_id\tpage\tuser_id\ns-t\t${id}\t/tsv-names\tu-t\n`, + }); + expect(res.status).toBe(200); + expect(((await res.json()) as { succeeded: number }).succeeded).toBe(1); + await waitForCondition(async (signal) => { + const r = await chQuery( + `SELECT page FROM default.${T.clicks} WHERE event_id = '${id}'`, + signal, + ); + return r.length === 1 && r[0].page === "/tsv-names"; + }, 10_000); + + // A header ClickHouse refuses is a verdict on the body, not on a record: the + // whole request fails with the clickhouse.rejected class and its code. + const bad = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { ...auth, "Content-Type": "text/csv; header=present" }, + body: `event_id,nosuch\n${testId()},1\n`, + }); + expect(bad.status).toBe(400); + const err = (await bad.json()) as { error?: string; code?: string; exception_code?: number }; + expect(err.code).toBe("clickhouse.rejected"); + expect(err.exception_code).toBe(117); + expect(err.error).toContain("nosuch"); + }); + + // CONTRACT CHANGE: an `_in` check has no single value to inject, + // so a record omitting the column is judged on the TABLE's own default rather + // than failing closed on absence. country DEFAULTs to 'US' here, so a token + // whose allowed set contains 'US' admits the record and one without it does + // not — the old behaviour refused both. + it("tests the table default when an _in check column is absent", async () => { + const currentPolicy = readPolicyFile(); + // `_in` takes a claim TEMPLATE, not a literal set: the allowed values come + // from the token, which is the multi-tenant case the check exists for. + await setPolicy({ + tables: { + ...currentPolicy.tables, + [T.clicks]: { + ...(currentPolicy.tables[T.clicks] || {}), + viewer: { + ...(currentPolicy.tables[T.clicks]?.viewer || {}), + insert: { allow_columns: ["*"], check: { country: { _in: "{{ jwt.countries }}" } } }, + }, + }, + }, + }); + + const post = async (id: string, countries: string[]) => + fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { + Authorization: `Bearer ${makeJWT({ sub: "test", role: "viewer", countries })}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + event_id: id, + page: "/in-default", + user_id: "u-in", + session_id: "s-in", + }), + }); + + try { + // country DEFAULTs to 'US'; the token allows it, so the record is admitted + // and stored with that default — the behaviour change. + const okId = testId(); + const ok = await post(okId, ["US", "CA"]); + expect(ok.status).toBe(200); + await waitForCondition(async (signal) => { + const r = await chQuery( + `SELECT country FROM default.${T.clicks} WHERE event_id = '${okId}'`, + signal, + ); + return r.length === 1 && r[0].country === "US"; + }, 10_000); + + // The same absent column with a token that does NOT allow the default is + // refused — the check is really being evaluated, not skipped. + const badId = testId(); + const bad = await post(badId, ["CA", "MX"]); + expect(bad.status).toBe(403); + const body = (await bad.json()) as { error?: string }; + expect(body.error).toContain("check failed"); + } finally { + await setPolicy(currentPolicy); + } + }); + it("rejects invalid JSON queries", async () => { const res = await fetch(`${WH_URL}/v1/query?table=${T.clicks}`, { method: "POST", diff --git a/tests/e2e/sdk/ndjson.test.ts b/tests/e2e/sdk/ndjson.test.ts index 9ed8cf00..49d0d74e 100644 --- a/tests/e2e/sdk/ndjson.test.ts +++ b/tests/e2e/sdk/ndjson.test.ts @@ -122,7 +122,11 @@ describe("NDJSON ingest", () => { expect(result.data?.failed).toBe(1); const failed = result.data?.results?.find((r) => r.error); expect(failed?.index).toBe(2); - expect(failed?.error).toContain("invalid json"); + // CONTRACT CHANGE: the message is ClickHouse's own parse refusal, carrying + // its exception_code, where it used to be the Go decoder's flat + // "invalid json". + expect(typeof failed?.exception_code).toBe("number"); + expect(failed?.error).toBeTruthy(); await waitForCondition(async (signal) => { const r = await chQuery( @@ -133,6 +137,66 @@ describe("NDJSON ingest", () => { }, 10_000); }); + // The compact-array regression guard on the wire: a SINGLE-LINE JSON array + // with one bad record used to lose the whole batch — chtypes rejects it + // outright and exports no bytes, so the records that parsed perfectly went + // with it. + // Ingest rewrites the array's depth-1 commas to newlines in place, which + // restores #195's promise that one bad record never obscures the rest. + it("salvages the good records of a compact JSON array with one bad record", async () => { + const runId = testId(); + const good = [`${runId}-a`, `${runId}-c`]; + const body = + "[" + + [ + { event_id: good[0], page: "/a", user_id: `user-${runId}`, session_id: `s-${runId}` }, + { + event_id: `${runId}-b`, + page: "/b", + user_id: `user-${runId}`, + session_id: `s-${runId}`, + totally_fake_field: "nope", + }, + { event_id: good[1], page: "/c", user_id: `user-${runId}`, session_id: `s-${runId}` }, + ] + .map((r) => JSON.stringify(r)) + .join(",") + + "]"; + // Deliberately one line: JSON.stringify of the array would be too, but + // spelling it out is what makes the framing the subject of the test. + expect(body.includes("\n")).toBe(false); + + const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${makeJWT({ sub: "test-viewer", role: "viewer", tenant_id: "acme" })}`, + }, + body, + }); + expect(res.status).toBe(200); + const parsed = (await res.json()) as { + total: number; + succeeded: number; + failed: number; + results: Array<{ index: number; exception_code?: number }>; + }; + expect(parsed).toMatchObject({ total: 3, succeeded: 2, failed: 1 }); + expect(parsed.results[1].exception_code).toBe(117); + + await waitForCondition(async (signal) => { + const r = await chQuery<{ event_id: string }>( + `SELECT event_id FROM default.${T.clicks} WHERE user_id = 'user-${runId}'`, + signal, + ); + return r.length === 2; + }, 10_000); + const inCH = await chQuery<{ event_id: string }>( + `SELECT event_id FROM default.${T.clicks} WHERE user_id = 'user-${runId}'`, + ); + expect(inCH.map((r) => r.event_id).sort()).toEqual([...good].sort()); + }); + it("accepts a raw JSON array body (Content-Type: application/json) and lands every row", async () => { const runId = testId(); const rows = [1, 2].map((n) => ({ From a776b2ce5c4012a7ede3a1c843a48b8f0bdca05e Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:14:23 -0400 Subject: [PATCH 13/70] feat(app): open the type layer in API processes and bind it per tenant An API-role process opens one typelayer.Engine at boot (the clickhouse.chtypes_registry directory, else the SDK's search path) and refuses to start without an artifact; an ingest-only or sweeper-only process never opens it and boots with none installed. Each tenant's schema registry binds the tenant from every successful refresh, attached before its first one so a loaded registry is a bound one, and a registry a reload retires forgets it without waiting on anything. Only the tenant's current registry binds it: a refresh still in flight when its registry was retired is dropped, and one already binding forgets again afterwards, so a tenant a reload moved never keeps the previous database's tables and a removed tenant's do not come back. The engine is released after schema discovery, after the HTTP drain. The ingest handler judges with the engine, the stream hub evaluates row filters with stream.NewRowEvaluator over it, and pipes and structured queries cap their HTTP connections at the tenant's max_open_conns. Tests that boot the api role skip without the artifact, or fail under WAVEHOUSE_TEST_REQUIRE_CHTYPES=1; the test settings close the HTTP port too, so a ClickHouse on the developer's :8123 cannot answer them. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/app/app.go | 18 ++- internal/app/app_test.go | 50 +++++--- internal/app/dedupe_dynamodb_test.go | 6 +- internal/app/discoveries.go | 10 +- internal/app/mq_nats_test.go | 6 +- internal/app/roles_test.go | 46 ++++++- internal/app/types.go | 92 ++++++++++++++ internal/app/types_test.go | 171 +++++++++++++++++++++++++++ internal/app/wire.go | 53 ++++++++- 9 files changed, 418 insertions(+), 34 deletions(-) create mode 100644 internal/app/types.go create mode 100644 internal/app/types_test.go diff --git a/internal/app/app.go b/internal/app/app.go index 8dfb8eea..1a56feea 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -46,6 +46,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/settings" "github.com/Wave-RF/WaveHouse/internal/stream" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) // BuildInfo is the ldflags-stamped identity of the binary, served by @@ -101,6 +102,11 @@ type App struct { pools *chconn.Pools bootState *api.BootState discoveries *discoveries + // types is the type layer ingest judges records with and the hub + // evaluates row filters with, and bindings what keeps each tenant's + // compiled tables in step with its registry. API role only. + types *typelayer.Engine + bindings *typeBindings // dedup is one store per tenant, each following its own folder's switch, // and dedupeStats the figures of the one Pebble instance they share. dedup *dedupe.Stores @@ -177,8 +183,8 @@ func New(ctx context.Context, opts Options) (app *App, err error) { } slog.Info("process roles", "roles", a.cfg.Roles, "instance_id", a.cfg.InstanceID) // What each role wires; config.Validate refused a set these cannot serve. - // The API's discovery, dedupe, auth verifiers, hub bridge and keepalive - // wheel are per process: every API process runs its own. + // The API's type layer, discovery, dedupe, auth verifiers, hub bridge and + // keepalive wheel are per process: every API process runs its own. apiRole, ingestRole := a.cfg.Has(config.RoleAPI), a.cfg.Has(config.RoleIngest) if apiRole || ingestRole { if err := a.wireClickHouse(); err != nil { @@ -186,6 +192,9 @@ func New(ctx context.Context, opts Options) (app *App, err error) { } } if apiRole { + if err := a.wireTypes(); err != nil { + return nil, err + } a.wireDiscovery(ctx) if err := a.wireDedupe(ctx); err != nil { return nil, err @@ -345,3 +354,8 @@ func (a *App) Registry() *discovery.SchemaRegistry { // MQ is the broker, for a harness that publishes straight onto the ingest // queue. func (a *App) MQ() mq.Broker { return a.mq } + +// Types is the type layer, bound from every tenant's discovery, for a harness +// that evaluates rows the way the stream hub does; nil in a process without +// the api role. +func (a *App) Types() *typelayer.Engine { return a.types } diff --git a/internal/app/app_test.go b/internal/app/app_test.go index 7673a898..57338709 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -40,6 +40,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/testutil" "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) // None of these tests run in parallel: New installs a process-wide default @@ -68,9 +69,10 @@ func closedPort(t *testing.T) int { return tcp.Port } -// writeSettings materializes the embedded seed with the ClickHouse address -// pointed at a closed port, then applies patch to config.json's top-level -// blocks (each value re-marshaled whole). +// writeSettings materializes the embedded seed with the ClickHouse native and +// HTTP ports pointed at closed ones — the query paths speak HTTP, so a +// developer's ClickHouse on :8123 must not answer them — then applies patch to +// config.json's top-level blocks (each value re-marshaled whole). func writeSettings(t *testing.T, patch map[string]any) string { t.Helper() files, err := settings.Seed() @@ -80,6 +82,7 @@ func writeSettings(t *testing.T, patch map[string]any) string { var ch map[string]any require.NoError(t, json.Unmarshal(doc["clickhouse"], &ch)) ch["addr"] = closedAddr(t) + ch["http_port"] = closedPort(t) doc["clickhouse"], err = json.Marshal(ch) require.NoError(t, err) for key, val := range patch { @@ -132,12 +135,24 @@ func newApp(t *testing.T, cfg *config.Config, opts Options) *App { t.Helper() guardGlobals(t) opts.Config = cfg - a, err := New(t.Context(), opts) + a, err := newForTest(t.Context(), t, opts) require.NoError(t, err) t.Cleanup(func() { assert.NoError(t, a.Close(context.Background())) }) return a } +// newForTest is New for a test. A boot with the api role opens the type +// layer, which refuses to start without a chtypes artifact, so such a test is +// skipped where none is installed — or failed under +// WAVEHOUSE_TEST_REQUIRE_CHTYPES=1, as CI runs it. +func newForTest(ctx context.Context, t *testing.T, opts Options) (*App, error) { + t.Helper() + if opts.Config != nil && opts.Config.Has(config.RoleAPI) { + typelayer.SkipWithoutArtifact(t) + } + return New(ctx, opts) +} + func get(t *testing.T, h http.Handler, path string) *httptest.ResponseRecorder { t.Helper() rec := httptest.NewRecorder() @@ -159,6 +174,7 @@ func TestNew_DegradedBootServesDiagnostics(t *testing.T) { assert.NotNil(t, a.Registry()) assert.NotNil(t, a.MQ()) + assert.NotNil(t, a.Types()) assert.NoError(t, a.Close(context.Background())) assert.NoError(t, a.Close(context.Background()), "Close is idempotent") } @@ -388,7 +404,7 @@ func TestNew_NestedWithoutAnOperatorKeyWarnsTheOpsTreeIsClosed(t *testing.T) { logs := logtest.Capture(t, slog.LevelWarn) cfg := testConfig(t, settingsDir) cfg.Auth.OperatorKey = operatorKey - a, err := New(t.Context(), Options{Config: cfg}) + a, err := newForTest(t.Context(), t, Options{Config: cfg}) require.NoError(t, err) t.Cleanup(func() { assert.NoError(t, a.Close(context.Background())) }) return logs.String() @@ -562,7 +578,7 @@ func TestNew_RefusesALayerWithoutABackend(t *testing.T) { guardGlobals(t) cfg := testConfig(t, writeSettings(t, nil)) tc.unset(cfg) - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, tc.key+` "" has no wiring`) }) } @@ -585,7 +601,7 @@ func TestNew_DedupeOpenFailure(t *testing.T) { guardGlobals(t) cfg := testConfig(t, writeSettings(t, dedupeOn)) block(t, cfg.DataDir) - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, "dedupe open") }) t.Run("nested fails closed", func(t *testing.T) { @@ -622,7 +638,7 @@ func TestNew_QueueOpenFailure(t *testing.T) { guardGlobals(t) cfg := testConfig(t, writeSettings(t, nil)) block(t, cfg.DataDir, "DLQ_0") - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, "mq open") }) t.Run("nested costs the tenant alone", func(t *testing.T) { @@ -647,7 +663,7 @@ func TestNew_QueueSetupHonorsTheBootContext(t *testing.T) { guardGlobals(t) ctx, cancel := context.WithCancel(t.Context()) cancel() - _, err := New(ctx, Options{Config: testConfig(t, writeSettings(t, nil))}) + _, err := newForTest(ctx, t, Options{Config: testConfig(t, writeSettings(t, nil))}) require.ErrorIs(t, err, context.Canceled) require.ErrorContains(t, err, "mq open") } @@ -800,7 +816,7 @@ func TestNew_RedisCacheRefusesAnUnreadableTLSFile(t *testing.T) { guardGlobals(t) cfg := redisTestConfig(t, writeSettings(t, nil), closedAddr(t)) cfg.Cache.Redis.TLS = config.CacheRedisTLS{Enabled: true, CAFile: filepath.Join(t.TempDir(), "gone.pem")} - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, "cache init: cache.redis.tls.ca_file") } @@ -963,7 +979,7 @@ func TestNew_NestedLooseFileRefusesBoot(t *testing.T) { guardGlobals(t) root := writeNestedSettings(t, map[string]map[string]any{"acme": nil}) require.NoError(t, os.WriteFile(filepath.Join(root, "notes.txt"), []byte("scratch"), 0o600)) - a, err := New(t.Context(), Options{Config: testConfig(t, root)}) + a, err := newForTest(t.Context(), t, Options{Config: testConfig(t, root)}) require.Error(t, err) assert.Nil(t, a) assert.Contains(t, err.Error(), "settings directory") @@ -972,7 +988,7 @@ func TestNew_NestedLooseFileRefusesBoot(t *testing.T) { func TestNew_RefusesInvalidSettingsDirectory(t *testing.T) { guardGlobals(t) cfg := testConfig(t, t.TempDir()) // empty: every required file is missing - a, err := New(t.Context(), Options{Config: cfg}) + a, err := newForTest(t.Context(), t, Options{Config: cfg}) require.Error(t, err) assert.Nil(t, a) assert.Contains(t, err.Error(), "settings directory") @@ -1036,13 +1052,13 @@ func TestNew_LateBootFailureReleasesEverything(t *testing.T) { cfg := testConfig(t, dir) natsDir := filepath.Join(cfg.DataDir, "nats") require.NoError(t, os.WriteFile(natsDir, []byte("not a directory"), 0o600)) - a, err := New(t.Context(), Options{Config: cfg}) + a, err := newForTest(t.Context(), t, Options{Config: cfg}) require.Error(t, err) assert.Nil(t, a) assert.Contains(t, err.Error(), "mq open") require.NoError(t, os.Remove(natsDir)) - a, err = New(t.Context(), Options{Config: cfg}) + a, err = newForTest(t.Context(), t, Options{Config: cfg}) require.NoError(t, err, "the stores opened before the failure were released") assert.True(t, a.dedup.For(tenant.Default).Open()) assert.NoError(t, a.Close(context.Background())) @@ -1510,7 +1526,7 @@ func TestNew_RefusesAPoolAboveTheCeiling(t *testing.T) { guardGlobals(t) cfg := testConfig(t, writeSettings(t, poolSettings(closedAddr(t), 10))) cfg.ClickHouse.MaxTotalConns = 4 - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, "clickhouse.max_open_conns 10") require.ErrorContains(t, err, "clickhouse.max_total_conns 4") } @@ -1600,7 +1616,7 @@ func TestNew_NestedRefusesPoolsAboveTheCeiling(t *testing.T) { }) cfg := testConfig(t, root) cfg.ClickHouse.MaxTotalConns = 15 - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, "clickhouse.max_open_conns 10") require.ErrorContains(t, err, "at 20, above clickhouse.max_total_conns 15") } @@ -2033,7 +2049,7 @@ func TestClose_WaitsForALoopAReloadStopped(t *testing.T) { // Released on every way out, so a failed assertion leaves no loop stuck. release := sync.OnceFunc(func() { close(conn.release) }) defer release() - d := newDiscoveries(t.Context(), nil, func(tenant.ID, error) {}, func(tenant.ID) {}) + d := newDiscoveries(t.Context(), nil, func(tenant.ID, error) {}, func(tenant.ID) {}, nil) d.adopt("acme", discovery.NewSchemaRegistry(func() (driver.Conn, string) { return conn, "default" }, "acme", func(tenant.ID) time.Duration { return time.Hour })) loop := (*d.cur.Load())["acme"] diff --git a/internal/app/dedupe_dynamodb_test.go b/internal/app/dedupe_dynamodb_test.go index e9dd5290..f4bae653 100644 --- a/internal/app/dedupe_dynamodb_test.go +++ b/internal/app/dedupe_dynamodb_test.go @@ -175,7 +175,7 @@ func TestNew_DynamoDBDedupeTableMissing(t *testing.T) { guardGlobals(t) cfg := testConfig(t, writeSettings(t, dedupeOn)) dynamoConfig(t, cfg, false) - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, "dedupe open") require.ErrorContains(t, err, "ResourceNotFoundException") require.NotErrorIs(t, err, dedupe.ErrUnavailable) @@ -185,7 +185,7 @@ func TestNew_DynamoDBDedupeTableMissing(t *testing.T) { cfg := testConfig(t, dir) dynamoConfig(t, cfg, false) logs := bootLogged(t) - a, err := New(t.Context(), Options{Config: cfg}) + a, err := newForTest(t.Context(), t, Options{Config: cfg}) require.NoError(t, err) t.Cleanup(func() { assert.NoError(t, a.Close(context.Background())) }) assert.Contains(t, logs.String(), `level=ERROR msg="dedupe: dynamodb table is misconfigured`) @@ -367,7 +367,7 @@ func TestNew_DynamoDBDedupeRefusesNoRegion(t *testing.T) { cfg.Dedupe.DynamoDB.Region = "" t.Setenv("AWS_REGION", "") t.Setenv("AWS_DEFAULT_REGION", "") - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, "dynamodb region is not set") }) } diff --git a/internal/app/discoveries.go b/internal/app/discoveries.go index 1f86079e..b54305d9 100644 --- a/internal/app/discoveries.go +++ b/internal/app/discoveries.go @@ -37,6 +37,9 @@ type discoveries struct { // and onLoaded the first success: what /livez is driven by. onAttempt func(tenant.ID, error) onLoaded func(tenant.ID) + // onRetire, when set, is told each registry a reload retires, under mu + // and the reload lock: what it does must not wait on I/O. + onRetire func(tenant.ID, *discovery.SchemaRegistry) mu sync.Mutex // serializes reconcile, drop, adopt and close cur atomic.Pointer[map[tenant.ID]*tenantDiscovery] @@ -54,8 +57,8 @@ type tenantDiscovery struct { done chan struct{} } -func newDiscoveries(ctx context.Context, build func(tenant.ID, *settings.Store) *discovery.SchemaRegistry, onAttempt func(tenant.ID, error), onLoaded func(tenant.ID)) *discoveries { - d := &discoveries{ctx: ctx, build: build, onAttempt: onAttempt, onLoaded: onLoaded} +func newDiscoveries(ctx context.Context, build func(tenant.ID, *settings.Store) *discovery.SchemaRegistry, onAttempt func(tenant.ID, error), onLoaded func(tenant.ID), onRetire func(tenant.ID, *discovery.SchemaRegistry)) *discoveries { + d := &discoveries{ctx: ctx, build: build, onAttempt: onAttempt, onLoaded: onLoaded, onRetire: onRetire} d.cur.Store(&map[tenant.ID]*tenantDiscovery{}) return d } @@ -96,6 +99,9 @@ func (d *discoveries) reconcile(tenants *settings.Registry) { // the retired loops that have ended since. Under mu. func (d *discoveries) retire(td *tenantDiscovery) { td.cancel() + if d.onRetire != nil { + d.onRetire(td.id, td.registry) + } d.retired = append(slices.DeleteFunc(d.retired, func(r *tenantDiscovery) bool { select { case <-r.done: diff --git a/internal/app/mq_nats_test.go b/internal/app/mq_nats_test.go index ba519993..dd26e785 100644 --- a/internal/app/mq_nats_test.go +++ b/internal/app/mq_nats_test.go @@ -46,7 +46,7 @@ func TestNew_NATSBackend(t *testing.T) { _, ok := a.MQ().(*mq.ExternalNATS) require.True(t, ok, "mq.backend: nats wires mq.ExternalNATS, got %T", a.MQ()) assert.Equal(t, []string{ - "clickhouse", "schema discovery", "dedupe", "mq", "cache", "coord", + "clickhouse", "type layer", "schema discovery", "dedupe", "mq", "cache", "coord", "hub bridge", "keepalive", "ingest worker", "auth", "sighup", "settings watcher", "http server", }, componentNames(a)) @@ -102,7 +102,7 @@ func TestNew_NATSUnreachable(t *testing.T) { guardGlobals(t) cfg := natsConfig(t, "nats://"+closedAddr(t)) cfg.MQ.NATS.TopologyWait = time.Millisecond - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorIs(t, err, mq.ErrUnavailable) assert.ErrorContains(t, err, "mq open") } @@ -114,7 +114,7 @@ func TestNew_NATSTopologyMissing(t *testing.T) { guardGlobals(t) cfg := natsConfig(t, srv.URL()) cfg.MQ.NATS.TopologyWait = time.Millisecond - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorIs(t, err, mq.ErrTopology) assert.ErrorContains(t, err, "dead-letter stream") } diff --git a/internal/app/roles_test.go b/internal/app/roles_test.go index 55eceaac..e6536f1e 100644 --- a/internal/app/roles_test.go +++ b/internal/app/roles_test.go @@ -2,6 +2,7 @@ package app import ( "errors" + "fmt" "net" "net/http" "net/http/httptest" @@ -19,6 +20,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/settings" "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) // Each role wires its own components and nothing else; the settings registry, @@ -32,12 +34,12 @@ func TestNew_RolesChooseTheComponents(t *testing.T) { want []string }{ {"every role", config.AllRoles(), []string{ - "clickhouse", "schema discovery", "dedupe", "mq", "cache", "coord", + "clickhouse", "type layer", "schema discovery", "dedupe", "mq", "cache", "coord", "sweeper", "hub bridge", "keepalive", "ingest worker", "auth", "sighup", "settings watcher", "http server", }}, {"api", []config.Role{config.RoleAPI}, []string{ - "clickhouse", "schema discovery", "dedupe", "mq", "cache", "coord", + "clickhouse", "type layer", "schema discovery", "dedupe", "mq", "cache", "coord", "hub bridge", "keepalive", "auth", "sighup", "settings watcher", "http server", }}, @@ -66,11 +68,49 @@ func TestNew_RolesChooseTheComponents(t *testing.T) { } } +// Only the api role opens the type layer. Over a search path holding no +// chtypes artifact, a process without it boots — the ingest worker needs only +// the static insert settings — and an API process refuses to start, naming +// where it looked. +func TestNew_OnlyTheAPIRoleNeedsTheArtifact(t *testing.T) { + // No explicit directory, no $CHTYPES_REGISTRY, an empty per-user cache, + // and no fetch on demand. + t.Setenv("CHTYPES_REGISTRY", "") + t.Setenv("CHTYPES_AUTOFETCH", "") + t.Setenv("XDG_CACHE_HOME", t.TempDir()) + if _, err := typelayer.NewEngine(typelayer.Config{}); err == nil { + t.Skip("a chtypes artifact is installed in a system directory, so this host has no search path without one") + } + + for _, roles := range [][]config.Role{ + {config.RoleIngest}, + {config.RoleSweeper}, + {config.RoleIngest, config.RoleSweeper}, + } { + t.Run(fmt.Sprint(roles), func(t *testing.T) { + cfg := testConfig(t, writeSettings(t, nil)) + cfg.Roles = roles + a := newApp(t, cfg, Options{}) + assert.Nil(t, a.Types()) + assert.NotContains(t, componentNames(a), "type layer") + }) + } + + t.Run("api", func(t *testing.T) { + guardGlobals(t) + cfg := testConfig(t, writeSettings(t, nil)) + cfg.Roles = []config.Role{config.RoleAPI} + a, err := New(t.Context(), Options{Config: cfg}) + require.ErrorContains(t, err, "type layer: chtypes: no version artifacts on the registry search path") + assert.Nil(t, a) + }) +} + func TestNew_RefusesAConfigWithoutRoles(t *testing.T) { guardGlobals(t) cfg := testConfig(t, writeSettings(t, nil)) cfg.Roles = nil - _, err := New(t.Context(), Options{Config: cfg}) + _, err := newForTest(t.Context(), t, Options{Config: cfg}) require.ErrorContains(t, err, "roles is empty") } diff --git a/internal/app/types.go b/internal/app/types.go new file mode 100644 index 00000000..af56d76b --- /dev/null +++ b/internal/app/types.go @@ -0,0 +1,92 @@ +package app + +import ( + "sync" + "sync/atomic" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// binder is what typeBindings drives: *typelayer.Engine, or a test's +// recorder. +type binder interface { + Bind(id tenant.ID, serverVersion, serverTZ string, tables []*discovery.TableSchema) + Forget(id tenant.ID) +} + +// typeBindings ties each tenant's schema registry to the process's one type +// layer: every successful refresh of the registry rebinds the tenant's +// compiled tables, and the registry's retirement forgets them. +// +// A registry is retired under the reload lock while a refresh of it may be +// in flight — its loop is cancelled, not joined — and a refresh past its +// last query still runs its hooks. Unguarded, that late bind would bring a +// removed tenant's tables back for good, or, for a tenant a reload moved +// (#638), land the previous database's schema over the one its fresh +// registry bound. So each tenant has one owning registry, and only the +// owner's refreshes bind it. +type typeBindings struct { + eng binder + + mu sync.Mutex + tenants map[tenant.ID]*typeBinding +} + +// typeBinding is one tenant's: kept for the process lifetime, so the +// registries a tenant has over time share one bindMu. +type typeBinding struct { + // bindMu serializes the tenant's binds across its registries, so a + // retired registry's bind cannot interleave with its successor's. + bindMu sync.Mutex + // owner is the registry whose refreshes bind the tenant; nil once it is + // retired and before its successor is attached. + owner atomic.Pointer[discovery.SchemaRegistry] +} + +func newTypeBindings(eng binder) *typeBindings { + return &typeBindings{eng: eng, tenants: map[tenant.ID]*typeBinding{}} +} + +func (b *typeBindings) of(id tenant.ID) *typeBinding { + b.mu.Lock() + defer b.mu.Unlock() + tb, ok := b.tenants[id] + if !ok { + tb = &typeBinding{} + b.tenants[id] = tb + } + return tb +} + +// attach makes reg tenant id's owning registry and binds the tenant from +// each of its successful refreshes. Call it before reg's first refresh, so +// the boot refresh binds too. +func (b *typeBindings) attach(id tenant.ID, reg *discovery.SchemaRegistry) { + tb := b.of(id) + tb.owner.Store(reg) + reg.OnRefresh(func(serverVersion, serverTZ string, tables []*discovery.TableSchema) { + tb.bindMu.Lock() + defer tb.bindMu.Unlock() + if tb.owner.Load() != reg { + return // a refresh that outlived its registry's retirement + } + b.eng.Bind(id, serverVersion, serverTZ, tables) + if tb.owner.Load() != reg { + // Retired mid-bind: its Forget may have run before this bind + // created the tenant afresh. Forgotten again here, before a + // successor's first bind, which waits on bindMu. + b.eng.Forget(id) + } + }) +} + +// detach retires reg as tenant id's owner and forgets the tenant's compiled +// tables, when reg still owns it. It waits on nothing — a bind in flight +// finishes on its own and forgets after itself — so the reload hooks that +// retire registries stay free of I/O. +func (b *typeBindings) detach(id tenant.ID, reg *discovery.SchemaRegistry) { + if b.of(id).owner.CompareAndSwap(reg, nil) { + b.eng.Forget(id) + } +} diff --git a/internal/app/types_test.go b/internal/app/types_test.go new file mode 100644 index 00000000..0ee7bcff --- /dev/null +++ b/internal/app/types_test.go @@ -0,0 +1,171 @@ +package app + +import ( + "context" + "os" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/ClickHouse/clickhouse-go/v2/lib/driver" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil" +) + +// recordingBinder records what typeBindings asks the type layer to do, one +// line per call naming the tables a bind carried. during, when set, runs +// inside the next Bind. +type recordingBinder struct { + mu sync.Mutex + calls []string + during func() +} + +func (r *recordingBinder) Bind(id tenant.ID, _, _ string, tables []*discovery.TableSchema) { + r.mu.Lock() + call := "bind " + id.String() + for _, ts := range tables { + call += " " + ts.Name + } + r.calls = append(r.calls, call) + during := r.during + r.during = nil + r.mu.Unlock() + if during != nil { + during() + } +} + +func (r *recordingBinder) Forget(id tenant.ID) { + r.mu.Lock() + defer r.mu.Unlock() + r.calls = append(r.calls, "forget "+id.String()) +} + +func (r *recordingBinder) log() []string { + r.mu.Lock() + defer r.mu.Unlock() + return append([]string(nil), r.calls...) +} + +// registryOf is a registry of tenant id that discovers the one table named. +func registryOf(id tenant.ID, table string) *discovery.SchemaRegistry { + conn := testutil.NewSchemaConn([]*discovery.TableSchema{{Name: table, Columns: []discovery.Column{{Name: "id", Type: "String"}}}}) + return discovery.NewSchemaRegistry(func() (driver.Conn, string) { return conn, "default" }, id, + func(tenant.ID) time.Duration { return time.Hour }) +} + +// A registry binds its tenant from every successful refresh until it is +// retired; its retirement forgets the tenant, and a refresh that outlives it +// binds nothing. +func TestTypeBindings_TheOwnerBindsUntilRetired(t *testing.T) { + rec := &recordingBinder{} + b := newTypeBindings(rec) + reg := registryOf("acme", "a") + b.attach("acme", reg) + + require.NoError(t, reg.Refresh(t.Context())) + require.NoError(t, reg.Refresh(t.Context())) + b.detach("acme", reg) + require.NoError(t, reg.Refresh(t.Context())) + b.detach("acme", reg) + + assert.Equal(t, []string{"bind acme a", "bind acme a", "forget acme"}, rec.log()) +} + +// A registry retired while it binds — the reload that moved its tenant +// landing mid-refresh — forgets the tenant again once its bind returns, and +// its successor's first bind waits for that, so the previous database's +// tables never stay bound over the new one's. A late detach of the retired +// registry leaves the successor's binding alone. +func TestTypeBindings_RetiredMidBindNeverOutlivesTheSuccessor(t *testing.T) { + rec := &recordingBinder{} + b := newTypeBindings(rec) + old, successor := registryOf("acme", "old_db"), registryOf("acme", "new_db") + b.attach("acme", old) + + entered, release := make(chan struct{}), make(chan struct{}) + rec.during = func() { + close(entered) + <-release + } + oldDone := make(chan error, 1) + go func() { oldDone <- old.Refresh(context.Background()) }() + <-entered + + // The reload: retire the old registry, attach the new one, whose loop + // refreshes at once. + b.detach("acme", old) + b.attach("acme", successor) + successorDone := make(chan error, 1) + go func() { successorDone <- successor.Refresh(context.Background()) }() + select { + case <-successorDone: + t.Fatal("the successor bound while the retired registry's bind was in flight") + case <-time.After(50 * time.Millisecond): + } + + close(release) + require.NoError(t, <-oldDone) + require.NoError(t, <-successorDone) + b.detach("acme", old) + assert.Equal(t, []string{"bind acme old_db", "forget acme", "forget acme", "bind acme new_db"}, rec.log()) +} + +// Tenants bind independently: one tenant's retirement forgets only its own. +func TestTypeBindings_TenantsAreIndependent(t *testing.T) { + rec := &recordingBinder{} + b := newTypeBindings(rec) + acme, globex := registryOf("acme", "a"), registryOf("globex", "g") + b.attach("acme", acme) + b.attach("globex", globex) + require.NoError(t, acme.Refresh(t.Context())) + require.NoError(t, globex.Refresh(t.Context())) + b.detach("acme", acme) + require.NoError(t, globex.Refresh(t.Context())) + assert.Equal(t, []string{"bind acme a", "bind globex g", "forget acme", "bind globex g"}, rec.log()) +} + +// Every registry a reload retires is reported with its tenant: the one a +// drop retires (a tenant moved to another database) and the one a reconcile +// retires (a tenant no longer served). +func TestDiscoveries_ReportEveryRetiredRegistry(t *testing.T) { + type retired struct { + id tenant.ID + reg *discovery.SchemaRegistry + } + var got []retired + // Registries with no connection: their loops retry without ever loading. + build := func(id tenant.ID, _ *settings.Store) *discovery.SchemaRegistry { + return discovery.NewSchemaRegistry(func() (driver.Conn, string) { return nil, "" }, id, + func(tenant.ID) time.Duration { return time.Hour }) + } + d := newDiscoveries(t.Context(), build, func(tenant.ID, error) {}, func(tenant.ID) {}, + func(id tenant.ID, reg *discovery.SchemaRegistry) { got = append(got, retired{id, reg}) }) + t.Cleanup(func() { assert.NoError(t, d.close(context.Background())) }) + + root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) + tenants, _ := settings.Open(root) + require.NotNil(t, tenants) + d.reconcile(tenants) + acme, globex := d.For("acme"), d.For("globex") + require.NotNil(t, acme) + require.NotNil(t, globex) + require.Empty(t, got) + + d.drop([]tenant.ID{"acme", "initech"}) + require.Equal(t, []retired{{"acme", acme}}, got, "only a tenant with a registry has one to retire") + + require.NoError(t, os.RemoveAll(filepath.Join(root, "globex"))) + tenants.Reload("test") + d.reconcile(tenants) + assert.Equal(t, []retired{{"acme", acme}, {"globex", globex}}, got) + assert.NotNil(t, d.For("acme"), "the dropped tenant, still served, has a fresh registry") + assert.NotSame(t, acme, d.For("acme")) +} diff --git a/internal/app/wire.go b/internal/app/wire.go index e59660cf..b90868ea 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -36,6 +36,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/settings" "github.com/Wave-RF/WaveHouse/internal/stream" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) const ( @@ -347,7 +348,9 @@ func (a *App) wireClickHouse() error { // registry, see wireClickHouse): its pool's connection and the database that // pool was opened for — never the adopted document's, which a refused move // would pair with the pool the tenant kept, discovering a database its -// queries and inserts do not use. +// queries and inserts do not use. A tenant on no pool yields an untyped nil +// connection — never a nil *Manager inside a non-nil driver.Conn, which +// would pass a nil check and panic on use. func (a *App) discoverySource(id tenant.ID) discovery.Source { return func() (driver.Conn, string) { m := a.pools.For(id) @@ -372,6 +375,34 @@ func (a *App) registryFor(s *settings.Store) *discovery.SchemaRegistry { // per-call setting rather than a property of the pool it shares. func queryTimeout(s *settings.Store) time.Duration { return s.ClickHouse().QueryTimeout } +// maxOpenConns is the tenant's clickhouse.max_open_conns, which also caps the +// HTTP connections its pipes and structured queries hold. +func maxOpenConns(s *settings.Store) int { return s.ClickHouse().MaxOpenConns } + +// wireTypes opens the type layer: the process's one chtypes registry, which +// ingest judges every record with and the stream hub evaluates row filters +// with. Both are API work, so only an API process opens it — an ingest-only +// or sweeper-only process boots with no artifact installed, the worker +// needing only the static typelayer.InsertSettings. No artifact anywhere on +// the search path refuses boot: an API process could judge nothing. Opening +// reads manifests only; a ClickHouse line's library is opened by the first +// tenant bound to it, from that tenant's discovery (wireDiscovery), and a +// tenant whose line or zone this process cannot serve is unavailable on its +// own. Released after schema discovery, whose loops bind it, and so after +// the HTTP drain. +func (a *App) wireTypes() error { + eng, err := typelayer.NewEngine(typelayer.Config{RegistryDir: a.cfg.ClickHouse.ChtypesRegistry}) + if err != nil { + return fmt.Errorf("type layer: %w — an api-role process judges ingest and row filters with a chtypes artifact: install one (scripts/fetch-chtypes.sh), or name its directory in clickhouse.chtypes_registry", err) + } + a.types, a.bindings = eng, newTypeBindings(eng) + a.add(component{name: "type layer", close: func(context.Context) error { + eng.Close() + return nil + }}) + return nil +} + // wireDiscovery builds one schema registry per served tenant, each with a // refresh loop of its own (discoveries), and the boot state /livez reports: // 503 with the latest discovery failure while no tenant has completed a @@ -410,7 +441,10 @@ func (a *App) wireDiscovery(ctx context.Context) { } d := newDiscoveries(a.stopCtx, func(id tenant.ID, _ *settings.Store) *discovery.SchemaRegistry { - return discovery.NewSchemaRegistry(a.discoverySource(id), id, perTenant(a.tenants, (*settings.Store).SchemaRefreshInterval)) + reg := discovery.NewSchemaRegistry(a.discoverySource(id), id, perTenant(a.tenants, (*settings.Store).SchemaRefreshInterval)) + // Before the first refresh, so "loaded" implies "bound". + a.bindings.attach(id, reg) + return reg }, func(id tenant.ID, err error) { slog.Warn("schema discovery retry failed", "tenant", id, "error", err) @@ -433,7 +467,8 @@ func (a *App) wireDiscovery(ctx context.Context) { loaded = true slog.Info("schema discovery succeeded after retry, /livez now 200", "tenant", id) a.bootState.Set(nil) - }) + }, + a.bindings.detach) a.discoveries = d if nested { a.bootState.Set(noTenantLoaded) @@ -758,6 +793,9 @@ func (a *App) wireSweeper() { func (a *App) wireStreaming() { a.sseMetrics = stream.NewMetrics() a.hub = stream.NewHub(perTenant(a.tenants, (*settings.Store).Policy), a.discoveries.For, a.sseMetrics) + // Row filters are evaluated by ClickHouse's own parser and expression + // engine over the published row, the answer the query path gives. + a.hub.RowEvaluator = stream.NewRowEvaluator(a.types) a.tenants.AfterAdopt(func([]tenant.ID) { a.hub.Prune(a.served) }) // Hub bridge: MQ → broadcast to connected SSE clients. The Hub decodes and @@ -984,6 +1022,7 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { ingestHandler.Dedup = func(s *settings.Store) dedupe.Deduplicator { return a.dedup.For(s.Tenant()) } ingestHandler.DedupeSettings = (*settings.Store).DedupeFor ingestHandler.DedupeLease = a.cfg.Dedupe.Lease + ingestHandler.Types = a.types // Readiness pings every open pool at once and is ready at the first // answer: one tenant's ClickHouse outage is not the process's. @@ -999,8 +1038,14 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { closing := make(chan struct{}) streamHandler.Closing = closing + // Pipes and structured queries run over the tenant's HTTP target, where + // ClickHouse renders the rows itself, holding at most the tenant's + // max_open_conns connections to it between them. pipesHandler := api.NewPipesHandler(func(s *settings.Store) pipes.Source { return s }, (*settings.Store).Policy, a.chTargetFor, a.cache, queryTimeout) pipesHandler.Tenants = a.tenants + pipesHandler.MaxConns = maxOpenConns + structuredQueryHandler := api.NewStructuredQueryHandler(a.chTargetFor, a.cache, a.registryFor, (*settings.Store).Policy, (*settings.Store).TimestampBucketSeconds, queryTimeout, (*settings.Store).DefaultMaxRows) + structuredQueryHandler.MaxConns = maxOpenConns schemaHandler := api.NewSchemaHandler(a.registryFor) schemaHandler.Tenants = a.tenants @@ -1020,7 +1065,7 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { Schema: schemaHandler, DLQ: api.NewDLQHandler(a.mq), Pipes: pipesHandler, - StructuredQuery: api.NewStructuredQueryHandler(a.chTargetFor, a.cache, a.registryFor, (*settings.Store).Policy, (*settings.Store).TimestampBucketSeconds, queryTimeout, (*settings.Store).DefaultMaxRows), + StructuredQuery: structuredQueryHandler, AuthMW: authMW, Tenants: a.tenants, From df6f93d5198a1decbeebada6a131a8879950d0dd Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:15:46 -0400 Subject: [PATCH 14/70] build(chtypes): move to SDK go/v0.5.1 and the ClickHouse 26.8 line The 26.6 line gets no further chtypes builds, so the lock moves to the 26.8 LTS line: 26.8.15.10-lts, build b1790845279, on darwin-arm64, linux-amd64 and linux-arm64, written by the v0.5.1 CLI as lock schema 2 (every consumer of the lock is on v0.5.1). scripts/fetch-chtypes.sh fetches line 26.8 with the same SDK version as go.mod, and CI and the e2e orchestrator pin clickhouse/clickhouse-server:26.8.15.10, the patch the artifact is built from. ABI revision 6 and the abi6 cache path are unchanged. The 26.8 build fills a skipped row's VerdictCode/VerdictErr, reports RowsPassed 0 for a rejected batch and sets ExportDeclined on a zero-row refusal, where every 26.6 build does not. The type layer already gates on Outcome and reads ErrCode/ErrMsg, which are right on both; the test that pinned the old RowsPassed now pins only what Ingest must do, and the comments say which builds behave which way. TestNewEngine_OpensNoLibraryAtConstruction now shadows the test line itself and releases a table it did not expect to get, so a passing lookup fails the test instead of deadlocking Close. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- .github/actions/setup-env/action.yml | 2 +- .github/workflows/ci.yml | 4 ++-- chtypes.lock | 23 +++++++++++-------- go.mod | 2 +- go.sum | 4 ++-- internal/typelayer/checks_test.go | 34 +++++++++++++++------------- internal/typelayer/ingest.go | 10 +++++--- internal/typelayer/roletable.go | 2 +- internal/typelayer/roletable_test.go | 4 ++-- internal/typelayer/tenancy_test.go | 2 +- internal/typelayer/testing.go | 9 ++++---- internal/typelayer/typelayer_test.go | 12 +++++++--- scripts/fetch-chtypes.sh | 16 +++++++------ scripts/orchestrator/main.go | 6 ++--- 14 files changed, 74 insertions(+), 56 deletions(-) diff --git a/.github/actions/setup-env/action.yml b/.github/actions/setup-env/action.yml index 123d011c..a4cea5ee 100644 --- a/.github/actions/setup-env/action.yml +++ b/.github/actions/setup-env/action.yml @@ -198,7 +198,7 @@ runs: # # The SDK's default fetch dir is revision-scoped: # ~/.cache/chtypes/artifacts/abi/-. The path and the key - # prefix both name the ABI revision (6 at SDK v0.4.0), so a cache saved + # prefix both name the ABI revision (6 at SDK v0.5.1), so a cache saved # by an older SDK is never restored into the search path. When the SDK's # ABI revision changes, bump `abi6` in both places and re-lock (see # docs/development.md). diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f8b3407e..98898b8d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -241,7 +241,7 @@ jobs: # the same tag is safe — dockerd dedupes layer downloads). Image tag # must match tests/integration/setup_test.go. - name: Prefetch ClickHouse image (background) - run: docker pull -q clickhouse/clickhouse-server:26.6.3.62 >/tmp/clickhouse-pull.log 2>&1 & + run: docker pull -q clickhouse/clickhouse-server:26.8.15.10 >/tmp/clickhouse-pull.log 2>&1 & - id: setup uses: ./.github/actions/setup-env with: @@ -284,7 +284,7 @@ jobs: # the same tag is safe — dockerd dedupes layer downloads). Image tag # must match scripts/orchestrator/main.go. - name: Prefetch ClickHouse image (background) - run: docker pull -q clickhouse/clickhouse-server:26.6.3.62 >/tmp/clickhouse-pull.log 2>&1 & + run: docker pull -q clickhouse/clickhouse-server:26.8.15.10 >/tmp/clickhouse-pull.log 2>&1 & # Suffix is -e2e-cov, not -e2e: this cache must carry the # cover-instrumented server objects build-cover compiles below, but # saves are gated on an exact-key miss — the pre-existing -e2e entry diff --git a/chtypes.lock b/chtypes.lock index f65d419f..46b036b8 100644 --- a/chtypes.lock +++ b/chtypes.lock @@ -1,17 +1,20 @@ { - "schema": 1, + "schema": 2, "artifacts": { - "darwin-arm64/26.6": { - "file": "chtypes-26.6.8.7-stable-darwin-arm64-b1790767905.tar.gz", - "sha256": "11df5c31e638990c595670d6b92e9e8382a7216fcba433ae0f6fbe5a325ab8f6" + "darwin-arm64/26.8.15.10-lts": { + "file": "chtypes-26.8.15.10-lts-darwin-arm64-b1790845279.tar.gz", + "sha256": "eec0b34d810e285aba244aa3b869637f0c1fa14755d11f7f74c069571890bd43", + "abi_revision": 6 }, - "linux-amd64/26.6": { - "file": "chtypes-26.6.8.7-stable-linux-amd64-b1790767905.tar.gz", - "sha256": "2fcbe729ae110507af22d35a19c290c97c1fae9f036aa2183841331b8ec9770e" + "linux-amd64/26.8.15.10-lts": { + "file": "chtypes-26.8.15.10-lts-linux-amd64-b1790845279.tar.gz", + "sha256": "25182defe11b2932182bce50aec203e003f721c505cb684b9a6725eeec956a1e", + "abi_revision": 6 }, - "linux-arm64/26.6": { - "file": "chtypes-26.6.8.7-stable-linux-arm64-b1790767905.tar.gz", - "sha256": "c85effdb5512aac0e012304249b09c86b516f53b401613f251222267453351d3" + "linux-arm64/26.8.15.10-lts": { + "file": "chtypes-26.8.15.10-lts-linux-arm64-b1790845279.tar.gz", + "sha256": "a9513cbb2aca868ffd16c97f097a530bef5dc8d5c8c11433159a882179d12254", + "abi_revision": 6 } } } diff --git a/go.mod b/go.mod index e5839d38..54741a28 100644 --- a/go.mod +++ b/go.mod @@ -44,7 +44,7 @@ require ( github.com/samber/slog-sampling v1.7.0 github.com/stretchr/testify v1.12.1 github.com/testcontainers/testcontainers-go v0.44.0 - github.com/wave-rf/chtypes/go v0.4.0 + github.com/wave-rf/chtypes/go v0.5.1 go.opentelemetry.io/contrib/bridges/otelslog v0.20.1 go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.71.0 go.opentelemetry.io/contrib/instrumentation/runtime v0.71.0 diff --git a/go.sum b/go.sum index 60a4b185..f3ffb6a9 100644 --- a/go.sum +++ b/go.sum @@ -399,8 +399,8 @@ github.com/tklauser/numcpus v0.12.0 h1:NR85qdvHA9pFse3x3weVZ0r0ST8R6l5RHbZrlRaqo github.com/tklauser/numcpus v0.12.0/go.mod h1:ABHeXzJnr/qqwguhClkZKT1/8VABcYrsyUiUGobwWJg= github.com/vladopajic/go-test-coverage/v2 v2.18.7 h1:Kfpv8jWoC0muCAgk4bR3SvhNfqSQvO9iKZkrhdAZ4dM= github.com/vladopajic/go-test-coverage/v2 v2.18.7/go.mod h1:sTDv3QDUo3Vjo2azG9hyHfMlXH5ugnF1kqDr/4O3w4g= -github.com/wave-rf/chtypes/go v0.4.0 h1:bRNYUzNB+9MoPHmfbE0ebhHipbHPe04jW/7GClI7U4s= -github.com/wave-rf/chtypes/go v0.4.0/go.mod h1:gQE6FgwdtsXvpWlDr79q8XSwDmhiY1mIRa3gj3bOtgo= +github.com/wave-rf/chtypes/go v0.5.1 h1:0ke0Z2TY5yFljxMJcT9Zxmbu7QxaBeNU/bOcjoX6fQw= +github.com/wave-rf/chtypes/go v0.5.1/go.mod h1:gQE6FgwdtsXvpWlDr79q8XSwDmhiY1mIRa3gj3bOtgo= github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e h1:JVG44RsyaB9T2KIHavMF/ppJZNG9ZpyihvCd0w101no= github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e/go.mod h1:RbqR21r5mrJuqunuUZ/Dhy/avygyECGrLceyNeo4LiM= github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU= diff --git a/internal/typelayer/checks_test.go b/internal/typelayer/checks_test.go index 05042ee5..e732dd11 100644 --- a/internal/typelayer/checks_test.go +++ b/internal/typelayer/checks_test.go @@ -27,7 +27,7 @@ func checksTable() *discovery.TableSchema { } // checksBody is four records whose verdicts under the cases below were -// measured on the 26.6 artifact. +// measured on the 26.6 and 26.8 artifacts. const checksBody = `{"id":1,"tenant":"acme","kind":"a"}` + "\n" + `{"id":2,"tenant":"acme","kind":"a"}` + "\n" + `{"id":3,"tenant":"evil","kind":"a"}` + "\n" + @@ -236,9 +236,10 @@ func checksHandleFor(t *testing.T, eng *Engine, table string) *Table { // TestIngestChecks_ParseOutcomeDecidesFirst pins a measured trap: under the // compile profile's allow_errors_ratio a record that does not parse is // skipped, and chtypes answers it 'd' beside that outcome — with the verdict's -// own code and message EMPTY on 26.6. It must report its parse error (a 400 -// with code 27), never a check decline (a 422), and never shift a neighbour -// onto its answer. +// own code and message EMPTY on older artifact builds (every 26.6 build; the +// 26.8 build fills them). It must report its parse error (a 400 with code 27, +// read from ErrCode, which every build sets), never a check decline (a 422), +// and never shift a neighbour onto its answer. func TestIngestChecks_ParseOutcomeDecidesFirst(t *testing.T) { tbl := checksHandle(t) body := []byte(strings.Join([]string{ @@ -249,8 +250,8 @@ func TestIngestChecks_ParseOutcomeDecidesFirst(t *testing.T) { }, "\n") + "\n") preds := []Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}} - // The trap itself, at the SDK: a skipped row's verdict is 'd' with no - // verdict code, and its error is only in ErrCode/ErrMsg. + // The trap itself, at the SDK: a skipped row's verdict is 'd', and its + // error is in ErrCode/ErrMsg on every build. s := tbl.pool.first() expr, params, ok := tbl.render(preds) require.True(t, ok) @@ -283,11 +284,12 @@ func TestIngestChecks_ParseOutcomeDecidesFirst(t *testing.T) { } // TestIngestChecks_RejectedBatchExportsNothing pins the other measured trap: -// RowsPassed counts the admitted rows of a batch whose own outcome is -// rejected, and such a batch exports no bytes. It is reachable here — an -// NDJSON-declared single-line array is not reframed (only the JSON family's -// arrays are), and one bad element rejects it whole. Every record must be -// declined; none may be published on the strength of RowsPassed. +// on older artifact builds (every 26.6 build) RowsPassed counts the admitted +// rows of a batch whose own outcome is rejected (the 26.8 build reports 0), +// and such a batch exports no bytes. It is reachable here — an NDJSON-declared +// single-line array is not reframed (only the JSON family's arrays are), and +// one bad element rejects it whole. Every record must be declined; none may +// be published on the strength of RowsPassed. func TestIngestChecks_RejectedBatchExportsNothing(t *testing.T) { tbl := checksHandle(t) body := []byte(`[{"id":1,"tenant":"acme","kind":"a"},{"id":"x","tenant":"acme","kind":"a"},{"id":3,"tenant":"acme","kind":"a"}]`) @@ -300,8 +302,8 @@ func TestIngestChecks_RejectedBatchExportsNothing(t *testing.T) { chtypes.WithRowFilter(tbl.filterOn(s, expr, params))) require.NoError(t, err) require.Equal(t, chtypes.Rejected, res.Outcome) - require.Positive(t, res.RowsPassed, "the trap: an admitted row is counted in a rejected batch") require.Empty(t, res.Payload) + t.Logf("RowsPassed of the rejected batch: %d", res.RowsPassed) batch, err := tbl.Ingest(FormatJSONEachRow, body, preds...) require.NoError(t, err) @@ -393,10 +395,10 @@ func TestIngest_PositionalFormats(t *testing.T) { } // TestIngest_WithNamesFormats pins the header formats as measured on the 26.6 -// artifact: the header line is not a record (Rows index the data lines), it -// names the columns in any order, a column it omits takes its DEFAULT, and a -// name the schema lacks or a repeated name refuses the body whole with -// ClickHouse's own 117. +// and 26.8 artifacts: the header line is not a record (Rows index the data +// lines), it names the columns in any order, a column it omits takes its +// DEFAULT, and a name the schema lacks or a repeated name refuses the body +// whole with ClickHouse's own 117. func TestIngest_WithNamesFormats(t *testing.T) { tbl := checksHandle(t) diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go index 1ecfc854..439d751c 100644 --- a/internal/typelayer/ingest.go +++ b/internal/typelayer/ingest.go @@ -25,7 +25,8 @@ type IngestOptions struct { // FormatTSV (input_format_csv_detect_header / input_format_tsv_detect_header // = 0), so every line is a record. By default ClickHouse consumes a first // line that names the columns as a header (both settings are on by default; - // measured on the 26.6 server and artifact). Ignored for other formats. + // measured on the 26.6 and 26.8 servers and artifacts). Ignored for other + // formats. StrictPositional bool } @@ -165,7 +166,9 @@ func (t *Table) IngestWith(format Format, opts IngestOptions, body []byte, check } // Only a fully accepted batch exports bytes. Gate on the outcome, never on - // RowsPassed, which counts the admitted rows of a rejected batch too. + // RowsPassed or ExportDeclined: older artifact builds (every 26.6 build) + // count the admitted rows of a rejected batch and leave ExportDeclined + // empty when it rejects with no rows, and the outcome is right on both. if res.Outcome != chtypes.Accepted { if len(res.Rows) == 0 && res.Outcome == chtypes.Rejected && res.ErrCode != 0 && (format == FormatCSVWithNames || format == FormatTSVWithNames) { @@ -244,7 +247,8 @@ func rowVerdict(r chtypes.RowResult, line []byte, filtered bool) RowVerdict { return RowVerdict{Accepted: true, Line: line} case chtypes.Skipped, chtypes.Rejected: // ErrCode/ErrMsg, not VerdictCode/VerdictErr: chtypes answers such a row - // 'd', and on 26.6 leaves the verdict's own code and message empty. + // 'd', and older artifact builds (every 26.6 build) leave the verdict's + // own code and message empty, where ErrCode/ErrMsg are set on every build. return RowVerdict{Code: r.ErrCode, Message: r.ErrMsg} case chtypes.Unsupported, chtypes.AcceptedPoisoned: // AcceptedPoisoned holds a value no writer can honestly serialize, so diff --git a/internal/typelayer/roletable.go b/internal/typelayer/roletable.go index 0664defe..d035381b 100644 --- a/internal/typelayer/roletable.go +++ b/internal/typelayer/roletable.go @@ -37,7 +37,7 @@ const roleCacheSize = 256 // // Defaults maps a column to a literal value injected when a record omits it, // rendered into the column's DEFAULT clause. A value the record DOES supply -// still wins (measured on the 26.6 artifact; see +// still wins (measured on the 26.6 and 26.8 artifacts; see // TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue). Every key // must name a column the shape // keeps and must be an ordinary column (no DEFAULT, or a plain DEFAULT) — diff --git a/internal/typelayer/roletable_test.go b/internal/typelayer/roletable_test.go index 7588b4ab..dffbb46c 100644 --- a/internal/typelayer/roletable_test.go +++ b/internal/typelayer/roletable_test.go @@ -77,8 +77,8 @@ func TestRoleTable_DeniedColumnIsAbsentFromTheSchema(t *testing.T) { } // TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue, measured on -// the 26.6 artifact: DEFAULT '' fills a column the record omits, and a -// value the record DOES supply still wins. +// the 26.6 and 26.8 artifacts: DEFAULT '' fills a column the record +// omits, and a value the record DOES supply still wins. func TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue(t *testing.T) { eng := TestEngine(t, ordersTable()) tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"tenant": "acme"}}) diff --git a/internal/typelayer/tenancy_test.go b/internal/typelayer/tenancy_test.go index dd76fcc4..d6fa46e4 100644 --- a/internal/typelayer/tenancy_test.go +++ b/internal/typelayer/tenancy_test.go @@ -38,7 +38,7 @@ func unavailable(t *testing.T, eng *Engine, id tenant.ID) *Unavailable { return u } -// TestBind_TenantsOnOneLineWithDifferentZones: the process opened the 26.6 +// TestBind_TenantsOnOneLineWithDifferentZones: the process opened the test // line in UTC, so a second tenant on that line whose server reports another // zone cannot be served by this process. It alone is refused, with a cause // naming both zones; the first tenant, and a third in the line's own zone, diff --git a/internal/typelayer/testing.go b/internal/typelayer/testing.go index a626944c..56a98ea1 100644 --- a/internal/typelayer/testing.go +++ b/internal/typelayer/testing.go @@ -14,12 +14,13 @@ import ( "github.com/Wave-RF/WaveHouse/internal/tenant" ) -// TestServerVersion is the ClickHouse version TestEngine binds to — a real -// 26.6 patch release, so the registry resolves the 26.6 artifact line. -const TestServerVersion = "26.6.3.62" +// TestServerVersion is the ClickHouse version TestEngine binds to — the patch +// the locked 26.8 artifact is built from, so its line resolves to that +// artifact and the bound library is an exact match for the server. +const TestServerVersion = "26.8.15.10" // testLine is TestServerVersion's version line, the one artifact tests need. -const testLine = "26.6" +const testLine = "26.8" // RequireEnv set to "1" turns every artifact skip into a failure. CI sets it, // so a runner with a broken artifact cache fails loudly instead of quietly diff --git a/internal/typelayer/typelayer_test.go b/internal/typelayer/typelayer_test.go index f039b05e..60e3a6a2 100644 --- a/internal/typelayer/typelayer_test.go +++ b/internal/typelayer/typelayer_test.go @@ -389,18 +389,24 @@ func TestQuoteIdentifier_AgreesWithChsql(t *testing.T) { func TestNewEngine_OpensNoLibraryAtConstruction(t *testing.T) { TestEngine(t) // skips (or fails under WAVEHOUSE_TEST_REQUIRE_CHTYPES) without the real artifact + // The explicit directory is searched first, so its broken copy of the test + // line shadows the working one in the per-user cache. dir := t.TempDir() - line := filepath.Join(dir, "26.6") + line := filepath.Join(dir, testLine) require.NoError(t, os.MkdirAll(line, 0o750)) require.NoError(t, os.WriteFile(filepath.Join(line, "manifest.json"), - []byte(`{"library":"libchtypes.so","clickhouse_version":"26.6.8.7-stable","clickhouse_minor":"26.6"}`), 0o600)) + fmt.Appendf(nil, `{"library":"libchtypes.so","clickhouse_version":%q,"clickhouse_minor":%q}`, + TestServerVersion+"-lts", testLine), 0o600)) eng, err := NewEngine(Config{RegistryDir: dir}) require.NoError(t, err, "a broken artifact is not a construction error") t.Cleanup(eng.Close) eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) - _, err = eng.Table(tenant.Default, "events") + tbl, err := eng.Table(tenant.Default, "events") + if err == nil { + tbl.Release() // so the failure below is reported instead of Close deadlocking on it + } require.Error(t, err) require.True(t, IsUnavailable(err)) assert.Contains(t, err.Error(), line, "the SDK's message names the directory that failed") diff --git a/scripts/fetch-chtypes.sh b/scripts/fetch-chtypes.sh index a9695674..aa0d5427 100755 --- a/scripts/fetch-chtypes.sh +++ b/scripts/fetch-chtypes.sh @@ -14,7 +14,7 @@ # A line not in the lock, or a lock/registry mismatch, is a hard failure. # # Usage: scripts/fetch-chtypes.sh [ ...] [--platform ] [--dest ] -# ClickHouse minor line(s) to fetch, e.g. 26.6. Defaults to +# ClickHouse minor line(s) to fetch, e.g. 26.8. Defaults to # every line this repo needs today (LOCK_LINES below). # --platform os-arch pair to fetch for (default: host platform, chosen # by the SDK's own HostPlatform()). Pass linux-amd64 / @@ -43,7 +43,7 @@ if [ "$(go env CGO_ENABLED 2>/dev/null || echo 0)" != "1" ]; then fi # Bump together with go.mod's `require github.com/wave-rf/chtypes/go` line. -CHTYPES_SDK_VERSION="v0.4.0" +CHTYPES_SDK_VERSION="v0.5.1" CHTYPES_CLI="github.com/wave-rf/chtypes/go/cmd/chtypes@${CHTYPES_SDK_VERSION}" REPO_ROOT="$(cd "$(dirname "$0")/.." && pwd)" @@ -51,11 +51,13 @@ LOCK_FILE="${REPO_ROOT}/chtypes.lock" # Lines every deployment of this repo needs today. The e2e harness # (tests/integration/setup_test.go, scripts/orchestrator) and dev compose -# both pin ClickHouse 26.6.3.62; chtypes resolves by MINOR line (26.6), -# never nearest, so this is "26.6", not the exact patch. Widening -# this list is how a new line gets adopted: fetch it, add it here, commit -# the updated lock. -LOCK_LINES=(26.6) +# pin ClickHouse 26.8.15.10, the same patch as the locked 26.8.15.10-lts +# build. A line request installs the newest patch the lock pins for that +# line, and the gateway asks the registry for the line, never a nearest +# line, so this is "26.8", not the exact patch. Widening this list is how a +# new line gets adopted: fetch it with --lock, add it here, commit the +# updated lock. +LOCK_LINES=(26.8) platform="" dest="" diff --git a/scripts/orchestrator/main.go b/scripts/orchestrator/main.go index 10eacba8..479156d6 100644 --- a/scripts/orchestrator/main.go +++ b/scripts/orchestrator/main.go @@ -143,9 +143,9 @@ func run() error { log.Println("→ starting ClickHouse testcontainer (clean state per run)...") ch, err := testcontainers.GenericContainer(ctx, testcontainers.GenericContainerRequest{ ContainerRequest: testcontainers.ContainerRequest{ - // Pinned to match tests/integration/setup_test.go (26.8 changed - // numeric DateTime64 parsing; see the comment there). - Image: "clickhouse/clickhouse-server:26.6.3.62", + // Pinned to the patch chtypes.lock's artifact is built from, and to + // match tests/integration/setup_test.go. + Image: "clickhouse/clickhouse-server:26.8.15.10", ExposedPorts: []string{"9000/tcp", "8123/tcp"}, WaitingFor: wait.ForListeningPort("9000/tcp").WithStartupTimeout(60 * time.Second), Env: map[string]string{ From 3694f5e9b480e18e7b93b132cdb2f67aa245829c Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:16:05 -0400 Subject: [PATCH 15/70] fix(typelayer): set each library's zone through the SDK default chtypes.Timezone is superseded in SDK v0.5.1, and a direct write races the SDK's own read when another library opens. openLine now sets the zone with SetDefaultTimezone under the lock every open takes, so the zone an open reads is the one set for it. WithTimezone does not fit one registry built at boot, before any server has reported its zone, that then opens each line in the zone of its first tenant. The zone record is keyed on the opened library's path, the unit the SDK initialises once per process. A second registry that asks for a line already open in another zone gets the SDK's own, untyped refusal; it is recognised from the record and reported as the same per-tenant cause, with the SDK's words kept. The registry is asked for the server's line, as v0.4.0 resolved it, so every tenant on a line shares one library whichever patch its server runs. Which artifact answers a tenant is logged once each time the tenant's library changes, as a warning when the server runs another patch. Anything the SDK would print on stderr goes to the process log through its Progress writer. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/typelayer/typelayer.go | 44 +++++++- internal/typelayer/zone.go | 112 ++++++++++++++------ internal/typelayer/zone_test.go | 180 ++++++++++++++++++++++++++++++++ 3 files changed, 303 insertions(+), 33 deletions(-) create mode 100644 internal/typelayer/zone_test.go diff --git a/internal/typelayer/typelayer.go b/internal/typelayer/typelayer.go index 8a468648..ac6246cc 100644 --- a/internal/typelayer/typelayer.go +++ b/internal/typelayer/typelayer.go @@ -116,16 +116,38 @@ type tenantSet struct { // first Bind for that line, as that tenant's Unavailable. // // WithPreload is deliberately not used: it opens a library at construction, -// and chtypes.Timezone must be set before that from a server's own zone, -// which only discovery knows (see openLine). +// and a library's zone must come from a server's own zone, which only +// discovery knows (see openLine). func NewEngine(cfg Config) (*Engine, error) { - reg, err := chtypes.NewRegistry(cfg.RegistryDir, chtypes.WithAutoFetch(false)) + reg, err := chtypes.NewRegistry(cfg.RegistryDir, + chtypes.WithAutoFetch(false), + chtypes.WithFetchOptions(chtypes.FetchOptions{Progress: sdkLog{}})) if err != nil { return nil, err } return &Engine{reg: reg, tenants: make(map[tenant.ID]*tenantSet)}, nil } +// sdkLog carries what the SDK would otherwise print raw on stderr — its +// warning when a requested patch is not installed and another patch of the +// line answers instead — into the process log, one record per line. +type sdkLog struct{} + +func (sdkLog) Write(p []byte) (int, error) { + for line := range strings.SplitSeq(string(p), "\n") { + msg := strings.TrimPrefix(strings.TrimSpace(line), "chtypes: ") + if msg == "" { + continue + } + if warning, ok := strings.CutPrefix(msg, "WARNING: "); ok { + slog.Warn("chtypes SDK warning", "message", warning) + } else { + slog.Info("chtypes SDK", "message", msg) + } + } + return len(p), nil +} + // Table returns tenant id's current compiled handle for a table, read-locked. // The caller must Release it when done; the handle stays alive and stable for // the whole time it is held, so a concurrent rebind waits rather than pulling @@ -311,6 +333,9 @@ func (s *tenantSet) bind(serverVersion, serverTZ string, tables []*discovery.Tab } libChanged := s.lib != lib s.mu.RUnlock() + if libChanged { + logLibrary(s.id, serverVersion, lib) + } fresh := make([]pending, 0, len(tables)) keep := make(map[string]struct{}, len(tables)) @@ -381,6 +406,19 @@ func (s *tenantSet) bind(serverVersion, serverTZ string, tables []*discovery.Tab } } +// logLibrary records, once each time a tenant's library changes, which +// artifact answers for the tenant's server. A server on another patch of the +// line is answered by the line's artifact, and patches can differ. +func logLibrary(id tenant.ID, serverVersion string, lib *chtypes.Library) { + attrs := []any{"tenant", id, "server_version", serverVersion, "chtypes_version", string(lib.Version)} + if samePatch(lib.Version, serverVersion) { + slog.Info("chtypes library bound", attrs...) + return + } + slog.Warn("chtypes library bound from another patch of the server's line; verdicts follow the artifact's patch", + attrs...) +} + // Table is one compiled shape: a pool of identical schema handles plus the // caches built over them. Fields are written only under the exclusive lock a // rebind takes, so a holder of a Release-pending read lock sees a consistent diff --git a/internal/typelayer/zone.go b/internal/typelayer/zone.go index 8372d692..58bc807f 100644 --- a/internal/typelayer/zone.go +++ b/internal/typelayer/zone.go @@ -2,6 +2,7 @@ package typelayer import ( "fmt" + "strings" "sync" "github.com/wave-rf/chtypes/go/chtypes" @@ -9,44 +10,95 @@ import ( // This file is the one place the server-zone rule lives. // -// chtypes reads its Timezone package global once per library, when the library -// is first opened, and a library is opened at most once per process (the SDK -// dedupes by path). So a zone is a fact about this process and one ClickHouse -// version line, not about an Engine or a tenant: the first tenant to open a -// line fixes its zone, and a tenant on that line whose server reports another -// zone cannot be served by this process. Every library open goes through -// openLine, so the record below is complete. - -// lineZones is the zone each opened library was initialised with, keyed by the -// SDK's own *Library (one per opened artifact, shared by every Registry). -var lineZones = struct { - mu sync.Mutex - zone map[*chtypes.Library]string -}{zone: map[*chtypes.Library]string{}} - -// openLine resolves the library for serverVersion, opening it in tz when this -// process has not opened it yet. A non-empty cause is why the tenant cannot be -// served: no loadable artifact for the line (the SDK's own message), or a line -// already opened in another zone. +// chtypes initialises each library in one server timezone, once per process: +// the SDK opens an artifact path at most once and refuses to open it again in +// another zone, and a per-call session_timezone does not change how a bare +// DateTime is read. So a zone is a fact about this process and one opened +// library, not about an Engine or a tenant: the first tenant to open a line's +// library fixes its zone, and a tenant on that line whose server reports +// another zone cannot be served by this process. Every library open goes +// through openLine, so the record below is complete. +// +// The zone reaches the SDK through its process default (SetDefaultTimezone), +// not WithTimezone: a registry's WithTimezone is fixed when it is built, and +// the Engine's one registry is built at boot, before discovery has reported +// any server's zone, and then opens every line, each in the zone of the first +// tenant on it. The default is set under the same lock as every open, so the +// zone an open reads is the one set for it. + +// openedZones is the zone each opened library was initialised in, keyed by the +// library's path (the SDK opens one image per path, shared by every Registry). +var openedZones = struct { + mu sync.Mutex + lib map[string]openedLib +}{lib: map[string]openedLib{}} + +type openedLib struct { + line string + zone string +} + +// openLine resolves the library for serverVersion's line, opening it in tz +// when this process has not opened it yet. A non-empty cause is why the tenant +// cannot be served: no loadable artifact for the line (the SDK's own message), +// or a library already opened in another zone. func openLine(reg *chtypes.Registry, serverVersion, tz string) (*chtypes.Library, string) { - lineZones.mu.Lock() - defer lineZones.mu.Unlock() + line := versionLine(serverVersion) + + openedZones.mu.Lock() + defer openedZones.mu.Unlock() - chtypes.Timezone = tz // read only if this For is what opens the library - lib, err := reg.For(chtypes.Version(serverVersion)) + chtypes.SetDefaultTimezone(tz) // read only if this call is what opens the library + lib, err := reg.For(chtypes.Version(line)) if err != nil { + // The SDK refuses to open a path already open in another zone, which + // another Registry in this process (another Engine) reaches before it + // has resolved the line itself. Its error is untyped, so the case is + // recognised from the record, and its words are kept beside ours. + for _, o := range openedZones.lib { + if o.line == line && o.zone != tz { + return nil, zoneCause(tz, line, o.zone) + " (" + err.Error() + ")" + } + } return nil, err.Error() } - opened, seen := lineZones.zone[lib] + o, seen := openedZones.lib[lib.Path] if !seen { - lineZones.zone[lib] = tz + openedZones.lib[lib.Path] = openedLib{line: lib.Minor, zone: tz} return lib, "" } - if opened != tz { - return nil, fmt.Sprintf( - "ClickHouse reports server timezone %q, but this process opened the chtypes library for ClickHouse %s in %q; "+ - "one process serves one timezone per ClickHouse version line, so serve this tenant from another process or restart", - tz, lib.Minor, opened) + if o.zone != tz { + return nil, zoneCause(tz, lib.Minor, o.zone) } return lib, "" } + +func zoneCause(tz, line, opened string) string { + return fmt.Sprintf( + "ClickHouse reports server timezone %q, but this process opened the chtypes library for ClickHouse %s in %q; "+ + "one process serves one timezone per ClickHouse version line, so serve this tenant from another process or restart", + tz, line, opened) +} + +// versionLine is the ClickHouse line of a server version ("26.8.15.10" is +// "26.8"). The registry is asked for the line, never the patch: every tenant +// on a line shares one library whichever patch its server runs, and the line +// resolves to the same library for the life of the process. A spelling with no +// line is passed through for the SDK to refuse in its own words. +func versionLine(serverVersion string) string { + v := strings.TrimPrefix(strings.TrimSpace(serverVersion), "v") + major, rest, ok := strings.Cut(v, ".") + if !ok { + return serverVersion + } + minor, _, _ := strings.Cut(rest, ".") + return major + "." + minor +} + +// samePatch reports whether the loaded artifact was built from the server's +// own patch. An artifact's version names its channel ("26.8.15.10-lts"); a +// server's version() does not. +func samePatch(artifact chtypes.Version, serverVersion string) bool { + a, _, _ := strings.Cut(string(artifact), "-") + return a == strings.TrimPrefix(strings.TrimSpace(serverVersion), "v") +} diff --git a/internal/typelayer/zone_test.go b/internal/typelayer/zone_test.go new file mode 100644 index 00000000..f3efcb3c --- /dev/null +++ b/internal/typelayer/zone_test.go @@ -0,0 +1,180 @@ +package typelayer + +import ( + "encoding/json" + "fmt" + "log/slog" + "strings" + "sync/atomic" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/wave-rf/chtypes/go/chtypes" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" +) + +// logRecords decodes what logtest.Capture collected. +func logRecords(t *testing.T, buf *logtest.Buffer) []map[string]any { + t.Helper() + var out []map[string]any + for line := range strings.SplitSeq(strings.TrimRight(buf.String(), "\n"), "\n") { + if line == "" { + continue + } + var rec map[string]any + require.NoError(t, json.Unmarshal([]byte(line), &rec)) + out = append(out, rec) + } + return out +} + +// withMsg is the records whose message is msg. +func withMsg(recs []map[string]any, msg string) []map[string]any { + var out []map[string]any + for _, r := range recs { + if r["msg"] == msg { + out = append(out, r) + } + } + return out +} + +// TestBind_SDKZoneRefusalIsThatTenantsUnavailable: a second Engine (its own +// registry) asks for a line this process already opened in UTC, in another +// zone, before its registry has resolved the line. The SDK refuses the open +// itself, with an untyped error; the tenant gets the same zone cause the +// Engine's own check gives, with the SDK's words kept, and the first Engine's +// tenant keeps answering. +func TestBind_SDKZoneRefusalIsThatTenantsUnavailable(t *testing.T) { + first := TestEngine(t, eventsTable()) // the line is open in UTC from here on + + second, err := NewEngine(Config{}) + require.NoError(t, err) + t.Cleanup(second.Close) + second.Bind("tokyo", TestServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) + + u := unavailable(t, second, "tokyo") + assert.Empty(t, u.Table) + assert.Contains(t, u.Cause, `"Asia/Tokyo"`) + assert.Contains(t, u.Cause, `"UTC"`) + assert.Contains(t, u.Cause, "one timezone per ClickHouse version line") + assert.Contains(t, u.Cause, "already initialized", "the SDK's own refusal is kept beside ours") + + // The refusal is the tenant's, not the registry's: the same Engine serves a + // tenant in the line's zone. + second.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + answers(t, second, tenant.Default) + answers(t, first, tenant.Default) +} + +// TestBind_ZoneRecordIsKeyedOnTheLibrary: the record names the library the +// SDK opened, by path, with the zone it was opened in. +func TestBind_ZoneRecordIsKeyedOnTheLibrary(t *testing.T) { + eng := TestEngine(t, eventsTable()) + lib := boundLib(eng, tenant.Default) + require.NotNil(t, lib) + + openedZones.mu.Lock() + o, ok := openedZones.lib[lib.Path] + openedZones.mu.Unlock() + require.True(t, ok, "the bound library %s is recorded", lib.Path) + assert.Equal(t, openedLib{line: testLine, zone: "UTC"}, o) +} + +// TestBind_LogsTheLibraryOncePerTenant: which artifact answers for a tenant's +// server is logged when the tenant's library changes, not on every refresh, +// and a server on another patch of the line is a warning. +func TestBind_LogsTheLibraryOncePerTenant(t *testing.T) { + eng := TestEngine(t, eventsTable()) + logs := logtest.Capture(t, slog.LevelInfo) + + eng.Bind("exact", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind("exact", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + otherPatch := testLine + ".1.1" + eng.Bind("other", otherPatch, "UTC", []*discovery.TableSchema{eventsTable()}) + answers(t, eng, "other") + + recs := logRecords(t, logs) + exact := withMsg(recs, "chtypes library bound") + require.Len(t, exact, 1, "one record for the tenant's first bind, none for the refresh") + assert.Equal(t, "exact", exact[0]["tenant"]) + assert.Equal(t, TestServerVersion, exact[0]["server_version"]) + assert.Equal(t, TestServerVersion+"-lts", exact[0]["chtypes_version"]) + + other := withMsg(recs, "chtypes library bound from another patch of the server's line; verdicts follow the artifact's patch") + require.Len(t, other, 1) + assert.Equal(t, "WARN", other[0]["level"]) + assert.Equal(t, otherPatch, other[0]["server_version"]) +} + +// sdkWarnedPatches makes each run's patch request a pair the SDK has not +// warned about yet: it warns once per (requested, loaded) pair per process. +var sdkWarnedPatches atomic.Int64 + +// TestNewEngine_SDKWarningsGoToTheLog: the SDK writes its warning that a +// requested patch is not installed through the registry's Progress writer, +// which NewEngine points at the process log, so nothing reaches stderr raw. +// The Engine itself asks for lines, which never warn; this asks the registry +// for an uninstalled patch directly to make the SDK speak. +func TestNewEngine_SDKWarningsGoToTheLog(t *testing.T) { + eng := TestEngine(t) + logs := logtest.Capture(t, slog.LevelInfo) + + patch := fmt.Sprintf("%s.0.%d", testLine, sdkWarnedPatches.Add(1)) + openedZones.mu.Lock() + chtypes.SetDefaultTimezone("UTC") // the zone TestEngine opened the line in + res, err := eng.reg.Resolve(chtypes.Version(patch)) + openedZones.mu.Unlock() + require.NoError(t, err) + require.False(t, res.Exact) + + warnings := withMsg(logRecords(t, logs), "chtypes SDK warning") + require.Len(t, warnings, 1) + assert.Equal(t, "WARN", warnings[0]["level"]) + assert.Contains(t, warnings[0]["message"], patch) + assert.NotContains(t, warnings[0]["message"], "WARNING:", "the SDK's prefix is the record's level") +} + +func TestSDKLog_OneRecordPerLine(t *testing.T) { + logs := logtest.Capture(t, slog.LevelInfo) + + in := []byte("chtypes: WARNING: first\n\n==> narrative\n") + n, err := sdkLog{}.Write(in) + require.NoError(t, err) + assert.Equal(t, len(in), n) + + recs := logRecords(t, logs) + require.Len(t, recs, 2) + assert.Equal(t, "WARN", recs[0]["level"]) + assert.Equal(t, "first", recs[0]["message"]) + assert.Equal(t, "INFO", recs[1]["level"]) + assert.Equal(t, "==> narrative", recs[1]["message"]) +} + +func TestVersionLine(t *testing.T) { + t.Parallel() + for in, want := range map[string]string{ + "26.8.15.10": "26.8", + "26.8": "26.8", + "v26.8.15.10": "26.8", + "25.8.3.1-lts": "25.8", + "1.2.3.4": "1.2", + "26": "26", + "": "", + } { + assert.Equal(t, want, versionLine(in), "%q", in) + } +} + +func TestSamePatch(t *testing.T) { + t.Parallel() + assert.True(t, samePatch("26.8.15.10-lts", "26.8.15.10")) + assert.True(t, samePatch("26.8.15.10", "26.8.15.10")) + assert.False(t, samePatch("26.8.15.10-lts", "26.8.14.3")) + assert.False(t, samePatch("26.8.15.10-lts", "26.8")) +} From cd1ddd70a9ef3cdf74facb6d2079f7289c660933 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:16:05 -0400 Subject: [PATCH 16/70] fix(typelayer): compile with ClickHouse's type gates A table with a LowCardinality(UInt64) column exists on a server only because its CREATE passed allow_suspicious_low_cardinality_types, but the type layer compiled it without that gate, so chtypes refused every record with code 455 and every filter over the table declined. A FixedString wider than 256 and a Variant of similar types failed the same way with code 44. The compile profile now carries the ten type gates that the 25.8, 26.6 and 26.8 artifacts all accept. Every table compiled here already exists on the server, so admitting its types changes no verdict a server would give. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/typelayer/ingest_test.go | 67 +++++++++++++++++++++++++++++++ internal/typelayer/pool.go | 19 +++++++++ 2 files changed, 86 insertions(+) diff --git a/internal/typelayer/ingest_test.go b/internal/typelayer/ingest_test.go index ddc133c3..5b72e213 100644 --- a/internal/typelayer/ingest_test.go +++ b/internal/typelayer/ingest_test.go @@ -7,6 +7,8 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + "github.com/wave-rf/chtypes/go/chtypes" + "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/tenant" ) @@ -210,3 +212,68 @@ func TestCountRecords(t *testing.T) { assert.Equal(t, 2, countRecords([]byte("{}\n{}"), 0)) assert.Equal(t, 5, countRecords([]byte("{}\n{}"), 5), "chtypes' own count wins when it has one") } + +// gatedTable has one column of each type ClickHouse refuses to create unless a +// type gate is set. Such a table exists on a server only because its CREATE +// passed the gate there. +func gatedTable() *discovery.TableSchema { + return &discovery.TableSchema{ + Name: "gated", + Columns: []discovery.Column{ + {Name: "c", Type: "LowCardinality(UInt64)", Position: 1}, + {Name: "fs", Type: "FixedString(300)", Position: 2}, + {Name: "v", Type: "Variant(UInt32, Int64)", Position: 3}, + }, + } +} + +// TestIngest_TypeGatedColumnsInsertAndFilter: the compile profile carries the +// type gates, so a table with a LowCardinality(UInt64), a FixedString wider +// than 256 or a Variant of similar types accepts its records and answers its +// filters. The control compiles the same declarations without the gates: +// every record is refused (455 and 44 on the 26.6 and 26.8 artifacts), which +// is what this table got on every call before. +func TestIngest_TypeGatedColumnsInsertAndFilter(t *testing.T) { + eng := TestEngine(t, gatedTable()) + tbl, err := eng.Table(tenant.Default, "gated") + require.NoError(t, err) + defer tbl.Release() + + body := []byte(`{"c":5,"fs":"a","v":1}` + "\n" + `{"c":7,"fs":"b","v":2}` + "\n") + batch, err := tbl.Ingest(FormatJSONEachRow, body) + require.NoError(t, err) + require.Len(t, batch.Rows, 2) + for i, r := range batch.Rows { + assert.True(t, r.Accepted, "record %d: %d %s", i, r.Code, r.Message) + } + + isFive := Predicate{Column: "c", Op: "=", Values: []string{"5"}} + batch, err = tbl.Ingest(FormatJSONEachRow, body, isFive) + require.NoError(t, err) + assert.Equal(t, []string{"", ReasonFilter}, checkReasons(t, batch), "the insert check answers t, f") + + stored, err := tbl.Ingest(FormatJSONEachRow, body) + require.NoError(t, err) + for i, want := range []bool{true, false} { + row, err := tbl.ParseRow(tbl.WireColumns, stored.Rows[i].Line) + require.NoError(t, err) + visible, reason := row.VisibleWithReason([]Predicate{isFive}) + row.Close() + assert.Equal(t, want, visible, "stored row %d: %s", i, reason) + } + + for _, ts := range gatedTable().Columns { + ddl, err := tbl.lib.ReconstructDDL([]chtypes.DiscoveredColumn{{Name: ts.Name, Type: ts.Type, Position: 1}}) + require.NoError(t, err) + ungated, err := tbl.lib.CompileDDL(ddl, chtypes.WithCompileSettings(map[string]string{ + "input_format_allow_errors_ratio": "1", + "input_format_skip_unknown_fields": "0", + })) + require.NoError(t, err, ts.Type) + res, err := ungated.Rows(FormatJSONEachRow, []byte(`{"`+ts.Name+`":5}`+"\n"), InsertSettings()) + ungated.Close() + require.NoError(t, err, ts.Type) + assert.NotEqual(t, chtypes.Accepted, res.Outcome, "%s without the gates", ts.Type) + assert.Contains(t, []int{44, 455}, res.ErrCode, "%s without the gates: %s", ts.Type, res.ErrMsg) + } +} diff --git a/internal/typelayer/pool.go b/internal/typelayer/pool.go index 28f7cd98..17a9202e 100644 --- a/internal/typelayer/pool.go +++ b/internal/typelayer/pool.go @@ -40,9 +40,28 @@ func poolSize() int { // skip-and-continue so every record gets its own verdict; skip_unknown_fields=0 // makes an unknown field a real per-row rejection (ClickHouse code 117) instead // of silent data loss. Neither is ever forwarded to the real INSERT. +// +// The type gates admit the column types ClickHouse refuses to create by +// default. Every table compiled here already exists on the server, so its +// CREATE passed these gates there; without them chtypes refuses every record +// of, say, a LowCardinality(UInt64) table with code 455, which the server +// would insert. A gate name the loaded line does not know fails the compile +// with code 115 (allow_experimental_qbit_type does on 25.8), so the list holds +// only names measured accepted on 25.8, 26.6 and 26.8. var compileSettings = map[string]string{ "input_format_allow_errors_ratio": "1", "input_format_skip_unknown_fields": "0", + + "allow_suspicious_low_cardinality_types": "1", + "allow_suspicious_fixed_string_types": "1", + "allow_suspicious_variant_types": "1", + "allow_experimental_json_type": "1", + "allow_experimental_variant_type": "1", + "allow_experimental_dynamic_type": "1", + "allow_experimental_time_time64_type": "1", + "allow_experimental_bfloat16_type": "1", + "allow_experimental_object_type": "1", + "allow_experimental_nlp_functions": "1", } // schemaSlot is one compiled handle plus the filters compiled against it. A From 92c6c6e6af89c76f5c9e9b1ed76e0015713c08d7 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:19:13 -0400 Subject: [PATCH 17/70] test(integration): chtypes on the app harness, with policies and tokens Policy-driven tests now run against the wired app the way an operator drives it: withPolicy writes roles.json and policies.json into the settings directory and reloads it through the ops route, and bearer mints a token for the role under the app's JWT secret. default_role stays the admin role, so the rest of the suite is unchanged. Ported onto it: the stream-vs-query row-filter differential, every query now through the production /v1/query as its own role and every stream verdict from the hub's evaluator over the app's own type layer, with a check that the query side admits something; CSV, TSV and their WithNames forms landing in ClickHouse; a compact array with one bad record; an unknown header refused as 400 with exception_code 117; the published row being the stored row; filter values round-tripping both bindings; the type-rendering pin; a denied column refused per record with 117; an injected check column; an _in check judged on the table default; an integer claim that does not fit refused. The resource-cap test runs through the same harness, and a role's time cap is pinned through /v1/query against a slow view instead of through the native driver, which no read path uses now. Dropped with their subject: the Go row-filter narrowing and timestamp canonicalization differentials. The identifier fixtures quote names the way ClickHouse's backQuote does, control characters included. The suite's ClickHouse moves to 26.8.15.10. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- .../integration/identifier_roundtrip_test.go | 44 +- tests/integration/ingest_test.go | 294 ++++++++- tests/integration/query_binding_test.go | 84 +++ tests/integration/query_errors_test.go | 63 +- tests/integration/query_limits_test.go | 112 ++-- tests/integration/query_types_test.go | 102 ++++ tests/integration/rowfilter_narrowing_test.go | 224 ------- tests/integration/rowfilter_stream_test.go | 567 ++++++++++++++++++ tests/integration/setup_test.go | 146 ++++- .../integration/testdata/query_types_pin.json | 1 + .../timestamp_canonicalization_test.go | 163 ----- tests/integration/typelayer_wire_test.go | 171 ++++++ 12 files changed, 1444 insertions(+), 527 deletions(-) create mode 100644 tests/integration/query_binding_test.go create mode 100644 tests/integration/query_types_test.go delete mode 100644 tests/integration/rowfilter_narrowing_test.go create mode 100644 tests/integration/rowfilter_stream_test.go create mode 100644 tests/integration/testdata/query_types_pin.json delete mode 100644 tests/integration/timestamp_canonicalization_test.go create mode 100644 tests/integration/typelayer_wire_test.go diff --git a/tests/integration/identifier_roundtrip_test.go b/tests/integration/identifier_roundtrip_test.go index 9df59084..0adbd9fa 100644 --- a/tests/integration/identifier_roundtrip_test.go +++ b/tests/integration/identifier_roundtrip_test.go @@ -15,9 +15,10 @@ package tests // identifiers client-side or delegates to server-side {name:Identifier} params. // // Fixtures are created with chQuoteIdent — an INDEPENDENT quoter that mirrors -// ClickHouse's own backQuote() (escape backslash, then backtick, each with a -// backslash). Using a different routine than the production builder is the point: -// a bug in the builder cannot hide by "agreeing" with an identical bug here. +// ClickHouse's own backQuote() byte for byte (backslash and backtick escaped, +// and NUL, \b, \t, \n, \f, \r as their escape sequences). Using a different +// routine than the production builder is the point: a bug in the builder +// cannot hide by "agreeing" with an identical bug here. import ( "context" @@ -34,11 +35,39 @@ import ( ) // chQuoteIdent quotes a ClickHouse identifier the way the server's backQuote() -// does: escape backslash, then backtick, each with a leading backslash, and wrap -// in backticks. This is the test's trusted oracle for CREATING fixtures with -// arbitrary names — it is intentionally not the routine under test. +// does, so a fixture's DDL is what ClickHouse itself would write for the name: +// backslash and backtick get a leading backslash, the control characters +// backQuote spells as escapes become them, and the result is wrapped in +// backticks. This is the test's trusted oracle for CREATING fixtures with +// arbitrary names — it is intentionally not the routine under test, and is +// written byte by byte rather than as a replacer table so it shares no shape +// with it. func chQuoteIdent(name string) string { - return "`" + strings.NewReplacer(`\`, `\\`, "`", "\\`").Replace(name) + "`" + var b strings.Builder + b.WriteByte('`') + for i := 0; i < len(name); i++ { + switch c := name[i]; c { + case '\\', '`': + b.WriteByte('\\') + b.WriteByte(c) + case 0: + b.WriteString(`\0`) + case '\b': + b.WriteString(`\b`) + case '\t': + b.WriteString(`\t`) + case '\n': + b.WriteString(`\n`) + case '\f': + b.WriteString(`\f`) + case '\r': + b.WriteString(`\r`) + default: + b.WriteByte(c) + } + } + b.WriteByte('`') + return b.String() } // chQuoteString renders a ClickHouse single-quoted string literal for a @@ -134,6 +163,7 @@ func TestIntegration_WeirdColumnNamesRoundTrip(t *testing.T) { "tick`col", // embedded backtick `back\slash`, // embedded backslash — raw escaping drops it `dq"col`, // embedded double quote + "tab\tcol", // a control character backQuote writes as an escape } for i, col := range weirdCols { t.Run(col, func(t *testing.T) { diff --git a/tests/integration/ingest_test.go b/tests/integration/ingest_test.go index 183e059f..8b5cee95 100644 --- a/tests/integration/ingest_test.go +++ b/tests/integration/ingest_test.go @@ -14,6 +14,8 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/policy" ) // TestIngest_FlowsToClickHouseWithoutDLQ exercises the happy path: POST @@ -122,26 +124,292 @@ func TestIngest_ComputedColumns_FlowToClickHouse(t *testing.T) { // TestIngest_SuppliedComputedColumn_Rejected: the other half — a record that // names a computed column is refused at the API with a 400 naming it, rather -// than having the value silently dropped by the positional encoder. +// than having the value silently dropped by the positional encoder. The +// refusal is ClickHouse's own: a MATERIALIZED column is not one a record may +// carry, so the parser answers with its own message and code (117). func TestIngest_SuppliedComputedColumn_Rejected(t *testing.T) { - e := env(t) - table := createTable(t, "user_id String, digest String MATERIALIZED concat('d:', user_id)", "ORDER BY user_id", ) - resp, err := http.Post( - e.baseURL+"/v1/ingest?table="+url.QueryEscape(table), - "application/json", - strings.NewReader(`{"user_id":"dave","digest":"forged"}`), - ) + status, body := postIngest(t, table, "application/json", `{"user_id":"dave","digest":"forged"}`, "") + require.Equal(t, http.StatusBadRequest, status, "body=%v", body) + assert.Contains(t, body["error"], "digest") + assert.EqualValues(t, 117, body["exception_code"], "ClickHouse's own code, not a gateway guess") +} + +// postIngest is one POST to /v1/ingest with a verbatim body and an explicit +// Content-Type, as the suite's default admin caller or, with authorization +// set (bearer), as the role its token names. +func postIngest(t *testing.T, table, contentType, body, authorization string) (int, map[string]any) { + t.Helper() + req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, + env(t).baseURL+"/v1/ingest?table="+url.QueryEscape(table), strings.NewReader(body)) + require.NoError(t, err) + req.Header.Set("Content-Type", contentType) + if authorization != "" { + req.Header.Set("Authorization", authorization) + } + resp, err := http.DefaultClient.Do(req) require.NoError(t, err) defer resp.Body.Close() - require.Equal(t, http.StatusBadRequest, resp.StatusCode) + var decoded map[string]any + require.NoError(t, json.NewDecoder(resp.Body).Decode(&decoded)) + return resp.StatusCode, decoded +} - var body map[string]any - require.NoError(t, json.NewDecoder(resp.Body).Decode(&body)) - assert.Contains(t, body["error"], "digest") - assert.Contains(t, body["error"], "cannot be inserted") +// eventuallyRows waits out the ingest worker's batch window for the count of +// rows matching where to be want. A want of 0 holds only once something posted +// later has landed, so the tests asserting an absence post a record that +// lands after the refused ones and wait for it first. +func eventuallyRows(t *testing.T, table, where string, want uint64) { + t.Helper() + ctx := context.Background() + assert.Eventually(t, func() bool { + var count uint64 + err := env(t).chConn.QueryRow(ctx, + fmt.Sprintf("SELECT count() FROM %s WHERE %s", table, where)).Scan(&count) + return err == nil && count == want + }, 30*time.Second, 250*time.Millisecond, "expected %d row(s) in %s WHERE %s", want, table, where) +} + +// A single-line JSON array with one bad record: the other two still land. +// Measured on the artifact, an unframed array with one bad record makes the +// parser refuse the whole batch and export nothing, which would break the +// per-record promise of #195; ingest rewrites the array's depth-1 commas to +// newlines in place, which restores per-record salvage. Asserted in the table, +// not only in the response. +func TestIngest_CompactArray_OneBadRecord_TheOthersStillLand(t *testing.T) { + table := createTable(t, "user_id String, value UInt32", "ORDER BY user_id") + + status, body := postIngest(t, table, "application/json", + `[{"user_id":"a1","value":1},{"user_id":"a2","value":"not-a-number"},{"user_id":"a3","value":3}]`, "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + assert.EqualValues(t, 3, body["total"]) + assert.EqualValues(t, 2, body["succeeded"]) + assert.EqualValues(t, 1, body["failed"]) + + eventuallyRows(t, table, "user_id IN ('a1','a3')", 2) + eventuallyRows(t, table, "user_id = 'a2'", 0) +} + +// CSV is positional in the table's declaration order. The end-to-end +// assertion is what makes the positional contract real: a column-order bug +// would still answer 200. +func TestIngest_CSVBody_LandsInClickHouse(t *testing.T) { + table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") + + status, body := postIngest(t, table, "text/csv", "\"c1\",\"click\",7\n\"c2\",\"view\",9\n", "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + assert.EqualValues(t, 2, body["succeeded"]) + + eventuallyRows(t, table, "user_id = 'c1' AND event_type = 'click' AND value = 7", 1) + eventuallyRows(t, table, "user_id = 'c2' AND event_type = 'view' AND value = 9", 1) +} + +// A bare text/csv is ClickHouse's default CSV, so a first line spelling the +// column names is consumed as a header and only the data rows land. +func TestIngest_BareCSVHeader_LandsInClickHouse(t *testing.T) { + table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") + + status, body := postIngest(t, table, "text/csv", "user_id,event_type,value\n\"d1\",\"click\",7\n", "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + assert.EqualValues(t, 1, body["total"], "the detected header is not a record") + assert.EqualValues(t, 1, body["succeeded"]) + + eventuallyRows(t, table, "user_id = 'd1' AND value = 7", 1) + eventuallyRows(t, table, "user_id = 'user_id'", 0) +} + +// header=absent is strictly positional, so the same header line is one +// refused record and never reaches the table. +func TestIngest_CSVHeaderAbsent_LandsInClickHouse(t *testing.T) { + table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") + + status, body := postIngest(t, table, "text/csv; header=absent", "user_id,event_type,value\n\"a1\",\"click\",7\n", "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + assert.EqualValues(t, 2, body["total"]) + assert.EqualValues(t, 1, body["succeeded"]) + assert.EqualValues(t, 1, body["failed"]) + + eventuallyRows(t, table, "user_id = 'a1' AND value = 7", 1) + eventuallyRows(t, table, "user_id = 'user_id'", 0) +} + +// TSV is CSV's tab-separated twin, with a bad row beside a good one, so +// per-record salvage is covered for the positional formats too. +func TestIngest_TSVBody_LandsInClickHouse(t *testing.T) { + table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") + + status, body := postIngest(t, table, "text/tab-separated-values", "t1\tclick\t5\nt2\tview\tnope\n", "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + assert.EqualValues(t, 1, body["succeeded"]) + assert.EqualValues(t, 1, body["failed"]) + + eventuallyRows(t, table, "user_id = 't1' AND value = 5", 1) + eventuallyRows(t, table, "user_id = 't2'", 0) +} + +// text/csv; header=present addresses the columns by the header, in any order: +// the header is not a record, a column it omits takes the table's own +// DEFAULT, and a bad row is salvaged like any other. +func TestIngest_CSVWithNamesBody_LandsInClickHouse(t *testing.T) { + table := createTable(t, "user_id String, event_type String, value UInt32 DEFAULT 42", "ORDER BY user_id") + + status, body := postIngest(t, table, "text/csv; header=present", "event_type,user_id\nclick,h1\nview,h2\n", "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + assert.EqualValues(t, 2, body["total"], "the header line is not a record") + assert.EqualValues(t, 2, body["succeeded"]) + + eventuallyRows(t, table, "user_id = 'h1' AND event_type = 'click' AND value = 42", 1) + eventuallyRows(t, table, "user_id = 'h2' AND event_type = 'view' AND value = 42", 1) +} + +// The tab-separated twin of the header=present case, with a bad row beside a +// good one. +func TestIngest_TSVWithNamesBody_LandsInClickHouse(t *testing.T) { + table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") + + status, body := postIngest(t, table, "text/tab-separated-values; header=present", + "value\tuser_id\tevent_type\n5\tn1\tclick\nnope\tn2\tview\n", "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + assert.EqualValues(t, 1, body["succeeded"]) + assert.EqualValues(t, 1, body["failed"]) + + eventuallyRows(t, table, "user_id = 'n1' AND value = 5", 1) + eventuallyRows(t, table, "user_id = 'n2'", 0) +} + +// A header naming a column the table does not have is ClickHouse's own +// refusal of the body, before any record: a whole-request 400 carrying its +// code, and nothing stored. +func TestIngest_WithNamesUnknownHeader_Is400WithCode117(t *testing.T) { + table := createTable(t, "user_id String, value UInt32", "ORDER BY user_id") + + status, body := postIngest(t, table, "text/csv; header=present", "user_id,nosuch\nu1,1\n", "") + require.Equal(t, http.StatusBadRequest, status, "body=%v", body) + assert.Equal(t, "clickhouse.rejected", body["code"]) + assert.EqualValues(t, 117, body["exception_code"]) + assert.Contains(t, body["error"], "nosuch") + + status, body = postIngest(t, table, "application/json", `{"user_id":"after","value":2}`, "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + eventuallyRows(t, table, "user_id = 'after'", 1) + eventuallyRows(t, table, "1", 1) +} + +// A column the role may not write is not in the schema its records are +// parsed against, so a record naming one is ClickHouse's per-record refusal — +// 400 with code 117 — where it used to be the gateway's 403. The same role +// writing without it still lands, the column taking the table's default +// rather than any value of the caller's. +func TestIngest_DeniedColumn_IsClickHouseCode117(t *testing.T) { + table := createTable(t, "user_id String, secret String", "ORDER BY user_id") + withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{ + table: {"writer": {Insert: &policy.InsertPermissions{DenyColumns: []string{"secret"}}}}, + }}) + writer := bearer(t, "writer", nil) + + status, body := postIngest(t, table, "application/json", `{"user_id":"d1","secret":"leak"}`, writer) + require.Equal(t, http.StatusBadRequest, status, "body=%v", body) + assert.Contains(t, body["error"], "secret") + assert.EqualValues(t, 117, body["exception_code"]) + + status, body = postIngest(t, table, "application/json", `{"user_id":"d2"}`, writer) + require.Equal(t, http.StatusOK, status, "body=%v", body) + eventuallyRows(t, table, "user_id = 'd2' AND secret = ''", 1) + eventuallyRows(t, table, "user_id = 'd1'", 0) +} + +// An _eq check fills a record that omits its column from the claim, keeps a +// record's own value when it matches, and refuses one that does not (403, +// nothing stored). +func TestIngest_AutoInject_FillsAnAbsentCheckColumn(t *testing.T) { + table := createTable(t, "user_id String, tenant String", "ORDER BY user_id") + tmpl := "{{ jwt.tenant }}" + withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{ + table: {"writer": {Insert: &policy.InsertPermissions{Check: map[string]policy.Filter{"tenant": {Eq: &tmpl}}}}}, + }}) + writer := bearer(t, "writer", map[string]any{"tenant": "acme"}) + + status, body := postIngest(t, table, "application/json", `{"user_id":"i1"}`, writer) + require.Equal(t, http.StatusOK, status, "absent → injected; body=%v", body) + status, body = postIngest(t, table, "application/json", `{"user_id":"i2","tenant":"acme"}`, writer) + require.Equal(t, http.StatusOK, status, "supplied and matching; body=%v", body) + status, body = postIngest(t, table, "application/json", `{"user_id":"i3","tenant":"other"}`, writer) + require.Equal(t, http.StatusForbidden, status, "supplied and not matching; body=%v", body) + assert.Contains(t, body["error"], "check failed") + + eventuallyRows(t, table, "user_id = 'i1' AND tenant = 'acme'", 1) + eventuallyRows(t, table, "user_id = 'i2' AND tenant = 'acme'", 1) + eventuallyRows(t, table, "user_id = 'i3'", 0) +} + +// An _in check has no single value to inject, so a record omitting its column +// is judged on the TABLE's default rather than refused for the absence. Both +// directions, because the outcome is entirely what the default happens to be. +func TestIngest_CheckIn_AbsentColumn_TestsTheTableDefault(t *testing.T) { + tmpl := "{{ jwt.tenants }}" + policyFor := func(table string) policy.Policy { + return policy.Policy{Tables: map[string]policy.TablePolicy{ + table: {"writer": {Insert: &policy.InsertPermissions{Check: map[string]policy.Filter{"tenant": {In: &tmpl}}}}}, + }} + } + claims := map[string]any{"tenants": []any{"acme", "globex"}} + + t.Run("a default outside the set is refused", func(t *testing.T) { + table := createTable(t, "user_id String, tenant String", "ORDER BY user_id") + withPolicy(t, policyFor(table)) + writer := bearer(t, "writer", claims) + status, body := postIngest(t, table, "application/json", `{"user_id":"n1"}`, writer) + require.Equal(t, http.StatusForbidden, status, "body=%v", body) + assert.Contains(t, body["error"], "check failed") + status, body = postIngest(t, table, "application/json", `{"user_id":"n2","tenant":"globex"}`, writer) + require.Equal(t, http.StatusOK, status, "body=%v", body) + eventuallyRows(t, table, "user_id = 'n2'", 1) + eventuallyRows(t, table, "user_id = 'n1'", 0) + }) + + t.Run("a default inside the set is admitted", func(t *testing.T) { + table := createTable(t, "user_id String, tenant String DEFAULT 'acme'", "ORDER BY user_id") + withPolicy(t, policyFor(table)) + status, body := postIngest(t, table, "application/json", `{"user_id":"n3"}`, bearer(t, "writer", claims)) + require.Equal(t, http.StatusOK, status, "body=%v", body) + eventuallyRows(t, table, "user_id = 'n3' AND tenant = 'acme'", 1) + }) +} + +// An _eq check on an integer column compares the claim through the strict +// round-trip cast. 2^64+5 does not fit a UInt64, yet a plain String binding +// wrapped it onto 5 in the check — and the injected DEFAULT wraps it onto 5 +// too — so a record used to land under tenant 5. Now the check refuses it +// (403) and nothing is published, whether the record omits the column or +// supplies the wrapped value itself; the same policy with a claim that fits +// lands as before. +func TestIngest_IntegerCheckClaimThatDoesNotFit_IsRefused(t *testing.T) { + table := createTable(t, "user_id String, tenant UInt64", "ORDER BY user_id") + tmpl := "{{ jwt.tenant }}" + withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{ + table: {"writer": {Insert: &policy.InsertPermissions{Check: map[string]policy.Filter{"tenant": {Eq: &tmpl}}}}}, + }}) + over := bearer(t, "writer", map[string]any{"tenant": "18446744073709551621"}) + fits := bearer(t, "writer", map[string]any{"tenant": "5"}) + + for _, rec := range []string{`{"user_id":"o1"}`, `{"user_id":"o2","tenant":5}`, `{"user_id":"o3","tenant":"18446744073709551621"}`} { + status, body := postIngest(t, table, "application/json", rec, over) + require.Equal(t, http.StatusForbidden, status, "%s: body=%v", rec, body) + assert.Contains(t, body["error"], "check failed", rec) + } + + status, body := postIngest(t, table, "application/json", `{"user_id":"i1"}`, fits) + require.Equal(t, http.StatusOK, status, "body=%v", body) + status, body = postIngest(t, table, "application/json", `{"user_id":"i2","tenant":5}`, fits) + require.Equal(t, http.StatusOK, status, "body=%v", body) + + // The admitted records were posted after the refused ones, so once they + // have landed a published refusal would have landed too. + eventuallyRows(t, table, "user_id IN ('i1', 'i2') AND tenant = 5", 2) + eventuallyRows(t, table, "user_id IN ('o1', 'o2', 'o3')", 0) + eventuallyRows(t, table, "1 = 1", 2) } diff --git a/tests/integration/query_binding_test.go b/tests/integration/query_binding_test.go new file mode 100644 index 00000000..56e53365 --- /dev/null +++ b/tests/integration/query_binding_test.go @@ -0,0 +1,84 @@ +//go:build integration + +package tests + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +// TestStructuredQuery_FilterValuesRoundTrip drives every filter value shape +// through the production /v1/query, for both the `eq` (scalar `{pN:String}`) +// and `in` (`{pN:Array(String)}`) bindings. +// +// It exists because the two bindings need DIFFERENT encodings and a Go-side +// unit test can only assert what the author believed: a scalar parameter is +// read by ClickHouse's escaped-text reader (an unencoded backslash silently +// becomes an escape sequence, an unencoded tab or newline is a hard code-457 +// parse error), while an Array(String) element is read as a quoted literal (a +// raw tab rides through, but a quote or backslash must be escaped). Applying +// either encoding to the other's value is silent data loss, so the server is +// the oracle: seed the value, filter for it, and require exactly the one row. +func TestStructuredQuery_FilterValuesRoundTrip(t *testing.T) { + e := env(t) + table := createTable(t, "id String, v String", "ORDER BY id") + + values := map[string]string{ + "plain": "hello", + "single-quote": "it's", + "backslash": `a\b`, + "windows-path": `C:\Users\x`, + "tab": "a\tb", + "newline": "a\nb", + "carriage-return": "a\rb", + "literal-bs-n": `a\nb`, + "double-backslash": `a\\b`, + "percent": "100%", + "ampersand": "a&b", + "unicode": "héllo→", + "quote-and-slash": `it's a\b`, + "sql-ish": `') OR 1=1 --`, + "empty": "", + } + + ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second) + defer cancel() + for id, v := range values { + require.NoError(t, e.chConn.Exec(ctx, + fmt.Sprintf("INSERT INTO `%s` (id, v) VALUES (?, ?)", table), id, v), "seed %q", id) + } + + query := func(t *testing.T, filter map[string]any) []map[string]any { + t.Helper() + filter["columns"] = []string{"id"} + body, err := json.Marshal(filter) + require.NoError(t, err) + got := postJSON(t, e.baseURL+"/v1/query?table="+table, string(body)) + require.Equal(t, http.StatusOK, got.status, "body: %s", got.raw) + var rows []map[string]any + require.NoError(t, json.Unmarshal([]byte(got.raw), &rows)) + return rows + } + + for id, v := range values { + t.Run("eq/"+id, func(t *testing.T) { + rows := query(t, map[string]any{"filters": []any{map[string]any{"column": "v", "op": "eq", "value": v}}}) + require.Len(t, rows, 1, "eq on %q must match exactly the row that holds it", v) + require.Equal(t, id, rows[0]["id"]) + }) + + t.Run("in/"+id, func(t *testing.T) { + // A second element that is nowhere in the table keeps the list a + // real list, so a bad separator would show up as a wrong count. + rows := query(t, map[string]any{"filters": []any{map[string]any{"column": "v", "op": "in", "value": []string{v, "\x00absent\x00"}}}}) + require.Len(t, rows, 1, "in on %q must match exactly the row that holds it", v) + require.Equal(t, id, rows[0]["id"]) + }) + } +} diff --git a/tests/integration/query_errors_test.go b/tests/integration/query_errors_test.go index d29660ef..fed6ecbc 100644 --- a/tests/integration/query_errors_test.go +++ b/tests/integration/query_errors_test.go @@ -15,36 +15,47 @@ import ( "testing" "time" - "github.com/ClickHouse/clickhouse-go/v2" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" "github.com/Wave-RF/WaveHouse/internal/app" - "github.com/Wave-RF/WaveHouse/internal/chconn" "github.com/Wave-RF/WaveHouse/internal/config" + "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" ) -// queryError is the error envelope a failed ClickHouse query answers with. +// queryError is a query's answer: the status and raw body, and the error +// envelope a failed ClickHouse query answers with. type queryError struct { status int retryAfter string + raw string Error string `json:"error"` Code string `json:"code"` Retryable *bool `json:"retryable"` } func postJSON(t *testing.T, url, body string) queryError { + t.Helper() + return postJSONAs(t, url, body, "") +} + +// postJSONAs is postJSON as the role an Authorization value names (bearer); +// "" is the suite's default caller. +func postJSONAs(t *testing.T, url, body, authorization string) queryError { t.Helper() req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, url, strings.NewReader(body)) require.NoError(t, err) req.Header.Set("Content-Type", "application/json") + if authorization != "" { + req.Header.Set("Authorization", authorization) + } resp, err := http.DefaultClient.Do(req) require.NoError(t, err) defer func() { _ = resp.Body.Close() }() raw, err := io.ReadAll(resp.Body) require.NoError(t, err) - got := queryError{status: resp.StatusCode, retryAfter: resp.Header.Get("Retry-After")} + got := queryError{status: resp.StatusCode, retryAfter: resp.Header.Get("Retry-After"), raw: string(raw)} if resp.StatusCode != http.StatusOK { require.NoError(t, json.Unmarshal(raw, &got), string(raw)) } @@ -86,7 +97,7 @@ func TestQueryErrors_CallerFault(t *testing.T) { require.NoError(t, e.chConn.Exec(context.Background(), fmt.Sprintf("ALTER TABLE %s DROP COLUMN page", table))) got := postJSON(t, e.baseURL+"/v1/query?table="+url.QueryEscape(table), `{"columns":["page"]}`) assertQueryError(t, got, http.StatusBadRequest, "clickhouse.rejected", false) - assert.Contains(t, got.Error, "code: 47") + assert.Contains(t, strings.ToLower(got.Error), "code: 47") }) } @@ -162,27 +173,25 @@ func TestQueryErrors_ClickHouseDown(t *testing.T) { } } -// TestQueryErrors_TimeCapReachesClickHouse pins the driver behaviour the -// role time cap depends on: with a context deadline over 1s, clickhouse-go -// overwrites max_execution_time with deadline+5s, so an overrun ends as a -// bare DeadlineExceeded; with no deadline the cap reaches ClickHouse, which -// reports TIMEOUT_EXCEEDED — the code /v1/query answers as the caller's. -func TestQueryErrors_TimeCapReachesClickHouse(t *testing.T) { +// TestQueryErrors_RoleTimeCapIsTheCallers: a role's max_execution_time +// reaches ClickHouse as the query's own setting — the HTTP reader sends one on +// every read — and a query that outruns it is the caller's: 400 +// clickhouse.limit_exceeded, not retryable, where an overrun of the tenant's +// query_timeout alone would read as ClickHouse's state. The slow table is a +// view that sleeps half a second per row; measured on 26.8, a 1 s cap ends it +// with TIMEOUT_EXCEEDED (159) where the uncapped query returns its four rows. +func TestQueryErrors_RoleTimeCapIsTheCallers(t *testing.T) { e := env(t) - capped := clickhouse.Context(context.Background(), clickhouse.WithSettings(clickhouse.Settings{"max_execution_time": 1})) - const slow = "SELECT sleep(2) SETTINGS function_sleep_max_microseconds_per_block = 3000000" - - withDeadline, cancel := context.WithTimeout(capped, 1500*time.Millisecond) - defer cancel() - err := e.chConn.Exec(withDeadline, slow) - require.Error(t, err) - _, hasCode := chconn.ExceptionCode(err) - assert.False(t, hasCode, "a deadline over 1s must still override the cap: %v", err) - - noDeadline, cancel2 := context.WithCancel(capped) - defer cancel2() - err = e.chConn.Exec(noDeadline, slow) - require.Error(t, err) - code, _ := chconn.ExceptionCode(err) - assert.Equal(t, int32(159), code, "%v", err) + ctx := context.Background() + view := fmt.Sprintf("it_slow_view_%d", tableCounter.Add(1)) + require.NoError(t, e.chConn.Exec(ctx, "CREATE VIEW "+view+" AS SELECT number + sleepEachRow(0.5) AS n FROM numbers(4)")) + t.Cleanup(func() { _ = e.chConn.Exec(context.Background(), "DROP VIEW IF EXISTS "+view) }) + require.NoError(t, e.registry.Refresh(ctx)) + withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{view: { + "capped": {Select: &policy.SelectPermissions{AllowColumns: []string{"*"}, MaxExecutionTime: 1000 /* ms */}}, + }}}) + + got := postJSONAs(t, e.baseURL+"/v1/query?table="+view, `{"columns":["n"]}`, bearer(t, "capped", nil)) + assertQueryError(t, got, http.StatusBadRequest, "clickhouse.limit_exceeded", false) + assert.Contains(t, got.Error, "TIMEOUT_EXCEEDED") } diff --git a/tests/integration/query_limits_test.go b/tests/integration/query_limits_test.go index 2d8555fe..cb682b8e 100644 --- a/tests/integration/query_limits_test.go +++ b/tests/integration/query_limits_test.go @@ -5,21 +5,14 @@ package tests import ( "context" "fmt" - "net/http" - "net/http/httptest" "strings" "testing" "time" - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" - "github.com/Wave-RF/WaveHouse/internal/api" - "github.com/Wave-RF/WaveHouse/internal/auth" - "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/policy" - "github.com/Wave-RF/WaveHouse/internal/settings" ) // TestStructuredQuery_ResourceCapsEnforcedServerSide is the executable proof @@ -31,12 +24,11 @@ import ( // so a rejection in the capped cases is attributable to the cap, not a broken // query. // -// It drives the real StructuredQueryHandler (not the shared admin-stamped -// server, which bypasses all policy) against the package's real ClickHouse, as -// a non-admin `viewer` whose policy carries the cap under test. (Server-wide -// resource backstops are ClickHouse's job — its settings profiles / quotas — -// not WaveHouse's, so there's nothing global to assert here; this proves the -// per-role caps that ARE WaveHouse's to enforce.) +// It runs through the production /v1/query, as roles whose policy carries the +// cap under test, with tokens for them. (Server-wide resource backstops are +// ClickHouse's job — its settings profiles / quotas — not WaveHouse's, so +// there's nothing global to assert here; this proves the per-role caps that +// ARE WaveHouse's to enforce.) func TestStructuredQuery_ResourceCapsEnforcedServerSide(t *testing.T) { e := env(t) @@ -54,74 +46,42 @@ func TestStructuredQuery_ResourceCapsEnforcedServerSide(t *testing.T) { } tests := []struct { - name string - perms policy.SelectPermissions // viewer's per-table select caps - wantStatus int - wantBodyHas string // substring required in the response body + role string + perms policy.SelectPermissions // the role's per-table select caps + wantBodyHas string // substring required in the response body }{ - { - // Control: identical query, no resource cap → full result set. - name: "no cap returns all rows", - perms: policy.SelectPermissions{AllowColumns: []string{"*"}}, - wantStatus: http.StatusOK, - wantBodyHas: `"row-24"`, - }, - { - // max_rows_to_read bounds rows SCANNED — the lever that stops a - // full-table scan. A 25-row scan blows past a cap of 1. ClickHouse - // error code 158 == TOO_MANY_ROWS (the native driver surfaces the - // numeric code, not the HTTP interface's symbolic suffix). - name: "per-role max_rows_to_read is enforced (code 158 TOO_MANY_ROWS)", - perms: policy.SelectPermissions{AllowColumns: []string{"*"}, MaxRowsToRead: 1}, - wantStatus: http.StatusBadRequest, - wantBodyHas: "code: 158", - }, - { - // max_memory_usage bounds peak query memory — the lever that stops - // a heavy aggregation from exhausting the box. A 1-byte cap is below - // the floor any query allocates. Code 241 == MEMORY_LIMIT_EXCEEDED. - // (ByteSize literal 1 == 1 byte.) - name: "per-role max_memory_usage is enforced (code 241 MEMORY_LIMIT_EXCEEDED)", - perms: policy.SelectPermissions{AllowColumns: []string{"*"}, MaxMemoryUsage: 1}, - wantStatus: http.StatusBadRequest, - wantBodyHas: "code: 241", - }, + // Control: identical query, no resource cap → full result set. + {role: "uncapped", perms: policy.SelectPermissions{AllowColumns: []string{"*"}}, wantBodyHas: `"row-24"`}, + // max_rows_to_read bounds rows SCANNED — the lever that stops a + // full-table scan. A 25-row scan blows past a cap of 1. ClickHouse + // error code 158 == TOO_MANY_ROWS. + {role: "rows_capped", perms: policy.SelectPermissions{AllowColumns: []string{"*"}, MaxRowsToRead: 1}, wantBodyHas: "code: 158"}, + // max_memory_usage bounds peak query memory — the lever that stops a + // heavy aggregation from exhausting the box. A 1-byte cap is below the + // floor any query allocates. Code 241 == MEMORY_LIMIT_EXCEEDED. + {role: "memory_capped", perms: policy.SelectPermissions{AllowColumns: []string{"*"}, MaxMemoryUsage: 1}, wantBodyHas: "code: 241"}, } - + grants := policy.TablePolicy{} for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - // Fresh handler + policy per case. Cache is nil so every case - // actually executes against ClickHouse (no cross-case cache hit - // masking enforcement). defaultMaxRows 0 falls back to the builder's - // constant. singleflight's zero value is ready to use. - p := &policy.Policy{ - AdminRole: "admin", - Tables: map[string]policy.TablePolicy{ - table: {"viewer": {Select: &tt.perms}}, - }, - } - h := api.NewStructuredQueryHandler( - func(*settings.Store) driver.Conn { return e.chConn }, nil, func(*settings.Store) *discovery.SchemaRegistry { return e.registry }, func(*settings.Store) *policy.Policy { return p }, func(*settings.Store) int { return 60 }, func(*settings.Store) time.Duration { return 30 * time.Second }, nil, - ) - - req := httptest.NewRequest(http.MethodPost, - "/v1/query?table="+table, strings.NewReader(`{"select_all":true}`)) - req = req.WithContext(auth.WithRole(req.Context(), "viewer")) - // The handler is served without the router, so the test stands in - // for TenantMW; the fixed getters above never read the store. - req = req.WithContext(api.WithStore(req.Context(), &settings.Store{})) - rec := httptest.NewRecorder() - - h.Handle(rec, req) + grants[tt.role] = policy.RolePermissions{Select: &tt.perms} + } + withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{table: grants}}) - body := rec.Body.String() - require.Equal(t, tt.wantStatus, rec.Code, - "unexpected status; body: %s", body) - assert.Contains(t, body, tt.wantBodyHas) - if tt.wantStatus != http.StatusOK { - // The role's own cap: the caller's, and not retried. - assert.Contains(t, body, `"code":"clickhouse.limit_exceeded","retryable":false`) + for _, tt := range tests { + t.Run(tt.role, func(t *testing.T) { + // A filter per role keeps the query text distinct, so no case is + // answered from the cache another case filled. + body := fmt.Sprintf(`{"select_all":true,"filters":[{"column":"page","op":"neq","value":%q}]}`, tt.role) + got := postJSONAs(t, e.baseURL+"/v1/query?table="+table, body, bearer(t, tt.role, nil)) + if tt.role == "uncapped" { + require.Equal(t, 200, got.status, got.Error) + assert.Contains(t, got.raw, tt.wantBodyHas) + return } + // The role's own cap: the caller's, and not retried. ClickHouse's + // message spells the code "Code: N." over HTTP. + assertQueryError(t, got, 400, "clickhouse.limit_exceeded", false) + assert.Contains(t, strings.ToLower(got.Error), tt.wantBodyHas) }) } } diff --git a/tests/integration/query_types_test.go b/tests/integration/query_types_test.go new file mode 100644 index 00000000..309fcda3 --- /dev/null +++ b/tests/integration/query_types_test.go @@ -0,0 +1,102 @@ +//go:build integration + +package tests + +import ( + "context" + "fmt" + "net/http" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +// queryTypesGolden is the pinned `/v1/query` response body for one row of +// every ClickHouse type family. It is a byte-for-byte pin, not a semantic +// one: the JSON *rendering* of a value is the endpoint's public contract, +// so a change from `"12.5"` to `12.5` — or a reordering of the keys — must +// show up as a failing test and be re-recorded deliberately. +// +// Regenerate with `WAVEHOUSE_UPDATE_PIN=1 make test-integration` (or +// `-run TestQuery_TypeRendering_Pin`) after an intentional contract change, +// and put the diff in the PR description. +const queryTypesGolden = "testdata/query_types_pin.json" + +// queryTypesDDL is one column per rendering family. Order matters: it is the +// order `SELECT *` projects, which is part of what the pin records. +const queryTypesDDL = ` + i8 Int8, i16 Int16, i32 Int32, i64 Int64, + u8 UInt8, u16 UInt16, u32 UInt32, u64 UInt64, + f32 Float32, f64 Float64, + dec Decimal(10, 2), dec64 Decimal64(3), + s String, ls LowCardinality(String), fs FixedString(4), + uu UUID, en Enum8('a' = 1, 'b' = 2), bl Bool, + ip4 IPv4, ip6 IPv6, + d Date, dt DateTime('UTC'), dt64 DateTime64(3, 'UTC'), + arr Array(String), m Map(String, UInt8), + nn Nullable(Int32), nnull Nullable(String)` + +// queryTypesRow is the single row, written as SQL literals so the values +// reach ClickHouse without passing through a driver's own type mapping — +// the pin must describe ClickHouse's storage, not clickhouse-go's encoder. +// i64/u64 sit past 2^53 so the pin also records how 64-bit integers are +// spelled; fs is shorter than its FixedString(4) so the NUL padding shows. +const queryTypesRow = `( + -8, -16, -32, -9007199254740993, + 8, 16, 32, 18446744073709551615, + 0.1, 0.1, + 12.50, 1.500, + 'hello', 'lc', 'ab', + toUUID('11111111-2222-3333-4444-555555555555'), 'a', true, + '10.0.0.1', '::1', + '2026-01-15', '2026-01-15 10:30:00', '2026-01-15 10:30:00.123', + ['a', 'b'], map('k', 7), + 42, NULL)` + +// TestQuery_TypeRendering_Pin snapshots the exact JSON `/v1/query` returns for +// one row of every ClickHouse type family, against the server the suite pins. +// +// It exists because the structured-query response is the SDK's data contract +// and nothing else asserts on the *spelling* of a value: the e2e tables are +// all String/UInt32/DateTime64, so a Decimal silently changing from a JSON +// number to a JSON string, or a FixedString from a string to a byte array, +// would reach consumers with a green suite. Any diff here is a breaking API +// change and belongs in the CHANGELOG. +// +// ClickHouse renders the body (FORMAT JSONEachRow under the reader's pinned +// output settings): keys in SELECT order, Decimal as a JSON number. Measured +// identical on 26.6.3.62 and 26.8.7.19. +func TestQuery_TypeRendering_Pin(t *testing.T) { + e := env(t) + + table := createTable(t, queryTypesDDL, "ORDER BY i8") + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + require.NoError(t, e.chConn.Exec(ctx, + fmt.Sprintf("INSERT INTO `%s` VALUES %s", table, queryTypesRow)), + "seed the one typed row") + + // The suite's default caller is the admin role, an unrestricted grant, so + // select_all stays a bare SELECT * and the projection is the DDL order. + got := postJSON(t, e.baseURL+"/v1/query?table="+table, `{"select_all":true}`) + require.Equal(t, http.StatusOK, got.status, "body: %s", got.raw) + body := strings.TrimRight(got.raw, "\n") + + if os.Getenv("WAVEHOUSE_UPDATE_PIN") == "1" { + require.NoError(t, os.MkdirAll(filepath.Dir(queryTypesGolden), 0o750)) + require.NoError(t, os.WriteFile(queryTypesGolden, []byte(body+"\n"), 0o600)) + t.Logf("pin updated: %s", queryTypesGolden) + return + } + + want, err := os.ReadFile(queryTypesGolden) + require.NoError(t, err, "read pin; regenerate with WAVEHOUSE_UPDATE_PIN=1") + require.Equal(t, strings.TrimRight(string(want), "\n"), body, + "the /v1/query type-rendering contract changed — re-record deliberately "+ + "with WAVEHOUSE_UPDATE_PIN=1 and document the diff") +} diff --git a/tests/integration/rowfilter_narrowing_test.go b/tests/integration/rowfilter_narrowing_test.go deleted file mode 100644 index 6186d559..00000000 --- a/tests/integration/rowfilter_narrowing_test.go +++ /dev/null @@ -1,224 +0,0 @@ -//go:build integration - -package tests - -import ( - "context" - "encoding/json" - "fmt" - "testing" - "time" - - "github.com/stretchr/testify/require" - - "github.com/Wave-RF/WaveHouse/internal/discovery" - "github.com/Wave-RF/WaveHouse/internal/policy" - "github.com/Wave-RF/WaveHouse/internal/stream" -) - -// TestRowFilterNumeric_DifferentialAgainstClickHouse pins the stream/query -// row-visibility agreement with ClickHouse itself as the oracle (the #381 -// review's storage-narrowing fail-open): for every numeric column shape × -// insertable payload × filter constant × operator, the in-memory verdict -// (policy.Evaluate → RowVisible over the pre-insert payload, specs built the -// way stream.Hub builds them) must equal what a structured query returns over -// the STORED row — `WHERE v ?` with the constant bound exactly as -// predicatesToSQL binds it. A constant ClickHouse rejects with a type error -// means the role reads no rows on the query path, so the stream must withhold -// too. Inserts go through the worker's exact HTTP surface (JSONEachRow), so -// storage narrowing — Float32/Float64 rounding, Decimal scale truncation — is -// ClickHouse's own, not a lookalike. -func TestRowFilterNumeric_DifferentialAgainstClickHouse(t *testing.T) { - shapes := []struct { - name string - ddl string - payloads []any - constants []string - // looseConstants are out-of-range spellings whose ClickHouse reading - // was measured to vary by pair on one release — mathematical promotion - // ('256' vs UInt8), or a width-boundary WRAP that compares against a - // different value than written (2^63 vs Int64 reads as −2^63). The - // stream refuses them all (the range gate), so verdicts legitimately - // diverge in the withholding direction; only the subset half of the - // guarantee is asserted here: the stream must never admit where SQL - // hides. - looseConstants []string - }{ - { - name: "uint64", - ddl: "UInt64", - payloads: []any{ - json.Number("16777217"), - json.Number("9007199254740992"), - json.Number("9007199254740993"), // 2^53+1: float64 would collapse it onto its neighbor - "12345678901234567890", // string-encoded (the JS-precision-loss escape hatch), > 2^63 - json.Number("0"), - }, - constants: []string{ - "16777216", "16777217", - "9007199254740992", "9007199254740993", - "12345678901234567890", - "1e3", // exponent spelling: ClickHouse's integer cast errors the query - "1.5", // fractional constant: same - "-5", // negative vs unsigned: measured cast error, role reads no rows — strict parity holds - }, - // Wide or width-boundary constants: ClickHouse's reading varies by - // pair (promotion vs wrap); the stream refuses — subset assertion. - looseConstants: []string{"18446744073709551616", "99999999999999999999999"}, - }, - { - name: "int64", - ddl: "Int64", - payloads: []any{json.Number("-5"), json.Number("9007199254740993")}, - constants: []string{ - "-4", "-5", "9007199254740992", - }, - // 2^63 vs Int64 was measured to WRAP (compares as −2^63) — the - // exact case the range gate exists for; subset assertion only. - looseConstants: []string{"9223372036854775808"}, - }, - { - name: "uint8", - ddl: "UInt8", - payloads: []any{json.Number("0"), json.Number("5"), json.Number("255")}, - constants: []string{ - "5", "255", - "-1", // negative vs unsigned: measured cast error — strict parity holds - }, - // Past the width, ClickHouse promotes and compares mathematically - // while the stream refuses — subset assertion only. - looseConstants: []string{"256", "300"}, - }, - { - name: "float32", - ddl: "Float32", - payloads: []any{ - json.Number("16777217"), // stores as 16777216 — the review repro - json.Number("16777218"), - json.Number("0.1"), - json.Number("1.5"), - }, - constants: []string{"16777216", "16777217", "0.1", "1.5", "2", "1e3"}, - }, - { - name: "float64", - ddl: "Float64", - payloads: []any{ - json.Number("9007199254740992"), - json.Number("9007199254740993"), // stores rounded: the storage domain collapses it - json.Number("0.1"), - }, - constants: []string{"9007199254740992", "9007199254740993", "0.1"}, - }, - { - name: "decimal_10_2", - ddl: "Decimal(10, 2)", - payloads: []any{ - json.Number("1.005"), // stores as 1.00 (truncation, not rounding) - json.Number("1.006"), - json.Number("1.02"), - json.Number("-1.005"), - }, - constants: []string{"1.005", "1.004", "1", "1.5", "-1", "1.50", "1e3"}, - looseConstants: []string{"999999999"}, // past Precision−Scale: promoted on the SQL side, refused here - }, - } - - ops := []string{"=", "!=", ">", "<"} - - for _, sh := range shapes { - t.Run(sh.name, func(t *testing.T) { - t.Parallel() - table := createTable(t, "id UInt32, v "+sh.ddl, "ORDER BY id") - spec := numericColumnSpec(t, sh.ddl) - - inserted := make(map[int]any, len(sh.payloads)) - for i, payload := range sh.payloads { - if err := insertBestEffort(t, table, map[string]any{"id": uint32(i), "v": payload}); err != nil { - // Un-storable payloads are the documented transient (DLQ) - // class, out of the parity claim — skip, on the record. - t.Logf("payload %v not insertable into %s (%v); skipping", payload, sh.ddl, err) - continue - } - inserted[i] = payload - } - require.NotEmpty(t, inserted, "corpus must contain insertable payloads") - - for id, payload := range inserted { - for _, constant := range sh.constants { - for _, op := range ops { - stream := streamVerdict(t, table, op, constant, payload, spec) - sql, sqlErr := storedVerdict(t, table, uint32(id), op, constant) - if stream != sql { - t.Errorf("%s: payload %v %s %q — stream says %v, ClickHouse says %v (query err: %v)", - sh.ddl, payload, op, constant, stream, sql, sqlErr) - } - } - } - for _, constant := range sh.looseConstants { - for _, op := range ops { - stream := streamVerdict(t, table, op, constant, payload, spec) - sql, sqlErr := storedVerdict(t, table, uint32(id), op, constant) - if stream && !sql { - t.Errorf("%s: payload %v %s %q — stream admits where ClickHouse hides (query err: %v)", - sh.ddl, payload, op, constant, sqlErr) - } - } - } - } - }) - } -} - -// numericColumnSpec builds the policy.ColumnSpec for a numeric ClickHouse type -// through the very mapping production uses (stream.NumericSpecOf, the same -// classifier and spec builder as stream.Hub's columnSpecs), so this oracle can -// never validate a mapping the Hub no longer applies. -func numericColumnSpec(t *testing.T, chType string) policy.ColumnSpec { - t.Helper() - st, ok := discovery.NumericStorageOf(chType) - require.True(t, ok, "corpus types must classify: %s", chType) - return policy.ColumnSpec{Kind: policy.ColumnNumeric, Numeric: stream.NumericSpecOf(st)} -} - -// streamVerdict resolves a one-operator literal filter through the full -// production path (Evaluate → RowVisible) and reports whether the stream -// would deliver the payload's event. -func streamVerdict(t *testing.T, table, op, constant string, payload any, spec policy.ColumnSpec) bool { - t.Helper() - f := policy.Filter{} - switch op { - case "=": - f.Eq = &constant - case "!=": - f.Neq = &constant - case ">": - f.Gt = &constant - case "<": - f.Lt = &constant - default: - t.Fatalf("unknown op %q", op) - } - p := &policy.Policy{Tables: map[string]policy.TablePolicy{ - table: {"r": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"v": f}}}}, - }} - perms := policy.Evaluate(p, "r", table, "select", nil) - require.True(t, perms.Allowed) - return perms.RowVisible(map[string]any{"v": payload}, map[string]policy.ColumnSpec{"v": spec}) -} - -// storedVerdict asks ClickHouse whether the stored row satisfies the predicate, -// with the constant bound as a positional parameter exactly like -// predicatesToSQL emits it. A query error (an exact-domain cast rejecting the -// constant's spelling) means the role reads no rows on that path. -func storedVerdict(t *testing.T, table string, id uint32, op, constant string) (bool, error) { - t.Helper() - ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) - defer cancel() - var cnt uint64 - q := fmt.Sprintf("SELECT count() FROM %s WHERE id = ? AND v %s ?", table, op) - if err := sharedEnv.chConn.QueryRow(ctx, q, id, constant).Scan(&cnt); err != nil { - return false, err - } - return cnt == 1, nil -} diff --git a/tests/integration/rowfilter_stream_test.go b/tests/integration/rowfilter_stream_test.go new file mode 100644 index 00000000..e913a735 --- /dev/null +++ b/tests/integration/rowfilter_stream_test.go @@ -0,0 +1,567 @@ +//go:build integration + +package tests + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "io" + "math/big" + "net/http" + "net/url" + "strconv" + "strings" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/policy" + "github.com/Wave-RF/WaveHouse/internal/stream" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" +) + +// TestRowFilterStream_DifferentialAgainstClickHouse pins the stream/query +// row-visibility agreement with ClickHouse itself as the oracle: for every +// column shape × insertable payload × filter constant × operator, the +// stream's verdict must equal what the production /v1/query returns over the +// SAME stored row, for a role whose row filter is that one predicate. A query +// ClickHouse rejects means the role reads no rows on the query path, so the +// stream must withhold too. +// +// Both surfaces run as production runs them. The policy is adopted from the +// settings directory, one role per (constant, operator) cell, and each query +// goes through the wired server with a token for its role. The stream side is +// the hub's own evaluator over the app's own type layer, bound by the app's +// refresh hook to what discovery found (server version, zone, tables), and it +// parses the positional JSONCompactEachRow line the ingest path publishes for +// the row, read back from the server rather than reconstructed here. +// +// On an integer column both surfaces compare a claim through the strict cast +// (chsql.StrictInt), so there is a second oracle as well: the admitted set +// must be the mathematically correct one — the comparison itself when the +// constant is the canonical spelling of a value the column can hold, and +// nothing at all otherwise. Parity alone could not catch an over-admit both +// surfaces share, and the plain String binding had one: a constant at or past +// 2^64 wrapped on every integer width. +// +// Parity is STRICT on every constant and every operator: there is no "the +// stream may be stricter" bucket and no excluded case. A filter value bound in +// a TYPED parameter used to disagree with the server — a sub-64-bit column's +// out-of-domain constant ("256" against UInt8) admitted where the server hides +// it, and a constant past a Decimal's precision could not be bound at all. +// Binding every value as {p:String} closed both, so a divergence appearing +// here is a regression, not a known gap. +func TestRowFilterStream_DifferentialAgainstClickHouse(t *testing.T) { + shapes := []diffShape{ + { + name: "uint64", + ddl: "UInt64", + intType: "UInt64", + payloads: []any{ + json.Number("16777217"), + json.Number("9007199254740993"), // 2^53+1: float64 would collapse it onto its neighbor + "12345678901234567890", // string-encoded (the JS-precision-loss escape hatch), > 2^63 + json.Number("0"), + }, + constants: []string{ + "16777216", "16777217", + "9007199254740992", "9007199254740993", + "12345678901234567890", + "0", + // Spellings and magnitudes the old Go comparator refused outright + // (it could not reproduce ClickHouse's own inconsistency): an + // exponent form, a fractional constant, a negative against an + // unsigned column, and two values past the column's width. Both + // sides now go through ClickHouse, so these are strict-parity + // cases rather than a "never admit where SQL hides" subset. + "1e3", "1.5", "-5", + "18446744073709551616", "99999999999999999999999", + }, + }, + { + name: "int64", + ddl: "Int64", + intType: "Int64", + payloads: []any{json.Number("-5"), json.Number("9007199254740993")}, + constants: []string{"-4", "-5", "9007199254740992", "9223372036854775808"}, + }, + { + name: "uint8", + ddl: "UInt8", + intType: "UInt8", + payloads: []any{json.Number("0"), json.Number("5"), json.Number("255")}, + // "256"/"300" were excluded while the stream bound integers through + // the widest integer type and answered `5 < '256'` true where the + // server answers false. Binding every value as {p:String} closed + // that, so they are strict-parity constants — this is the pin for it. + constants: []string{"5", "255", "0", "-1", "256", "300"}, + }, + // The integer widths, with every boundary that used to wrap. + intShape("uint8_bounds", "UInt8", "UInt8"), + intShape("uint32_bounds", "UInt32", "UInt32"), + intShape("uint64_bounds", "UInt64", "UInt64"), + intShape("int64_bounds", "Int64", "Int64"), + intShape("uint128_bounds", "UInt128", "UInt128"), + intShape("int128_bounds", "Int128", "Int128"), + intShape("uint256_bounds", "UInt256", "UInt256"), + intShape("int256_bounds", "Int256", "Int256"), + intShape("nullable_uint64_bounds", "Nullable(UInt64)", "UInt64"), + { + name: "float32", + ddl: "Float32", + payloads: []any{ + json.Number("16777217"), // stores as 16777216 — the review repro + json.Number("16777218"), + json.Number("0.1"), + json.Number("1.5"), + }, + constants: []string{"16777216", "16777217", "0.1", "1.5", "2", "1e3"}, + }, + { + name: "float64", + ddl: "Float64", + payloads: []any{ + json.Number("9007199254740992"), + json.Number("9007199254740993"), // stores rounded: the storage domain collapses it + json.Number("0.1"), + }, + constants: []string{"9007199254740992", "9007199254740993", "0.1"}, + }, + { + name: "decimal_10_2", + ddl: "Decimal(10, 2)", + payloads: []any{ + json.Number("1.005"), // stores as 1.00 (truncation, not rounding) + json.Number("1.02"), + json.Number("-1.005"), + }, + // "999999999" is past Precision−Scale. It used to be a + // "never admit where SQL hides" case because the constant could not + // be bound as that Decimal at all; as a String parameter it is read + // in the column's domain and agrees outright. + constants: []string{"1.005", "1.004", "1", "1.5", "-1", "1.50", "1e3", "999999999"}, + }, + { + name: "string", + ddl: "String", + // Byte ordering, not collation: "9" sorts after "100", which is + // exactly the leak a text fallback would cause on a numeric column + // and exactly the right answer on this one. + // + // The last five are the {p:String} escaping cases. A ClickHouse + // query parameter is read by an escaped-text reader on BOTH + // surfaces — the server's HTTP interface and the chtypes artifact's + // filter params — so an unencoded backslash arrives as an escape + // sequence and an unencoded tab or newline does not parse at all. + // Measured before chsql.EscapeStringParam was applied in + // typelayer.render(): the backslash rows answered false and the + // tab/newline/trailing-backslash rows declined, while this oracle + // (a natively-bound literal) answered true — the stream hid rows + // the query path returns. Strict parity here is the pin for that. + payloads: []any{"acme", "9", "100", "", `a\b`, "a\tb", "a\nb", `trail\`, "O'Brien"}, + constants: []string{"acme", "beta", "9", "100", "", `a\b`, "a\tb", "a\nb", `trail\`, "O'Brien"}, + }, + { + name: "datetime", + ddl: "DateTime", + // The policy author's zone-less spelling against the stored instant. + payloads: []any{"2026-06-21 04:00:00", "2026-06-21T04:00:01Z"}, + constants: []string{"2026-06-21 04:00:00", "2026-06-21 04:00:01"}, + }, + } + + // One row under test per (shape, payload): the table it lives in, its id, and + // the positional line the ingest path would publish for it. + type storedRow struct { + shape string + intType string + table string + id uint32 + payload any + columns []string + line []byte + } + var rows []storedRow + + // One policy for the whole corpus: on each shape's table, a role per + // (constant, operator) cell filtering v by that one predicate, and on an + // integer table one more whose _in reads a claim array. + tables := map[string]policy.TablePolicy{} + for _, sh := range shapes { + table := createTable(t, "id UInt32, v "+sh.ddl, "ORDER BY id") + grants := policy.TablePolicy{} + for i, constant := range sh.constants { + for _, op := range diffOps { + grants[cellRole(i, op)] = rowFilterGrant(t, op, constant) + } + } + if sh.intType != "" { + claimIn := "{{ jwt.ids }}" + grants[claimInRole] = policy.RolePermissions{Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"v": {In: &claimIn}}}} + } + tables[table] = grants + + anyStored := false + for i, payload := range sh.payloads { + if err := rowFilterInsert(t, table, map[string]any{"id": uint32(i), "v": payload}); err != nil { + // Un-storable payloads are the documented transient (DLQ) class, + // out of the parity claim — skip, on the record. + t.Logf("payload %v not insertable into %s (%v); skipping", payload, sh.ddl, err) + continue + } + anyStored = true + rows = append(rows, storedRow{ + shape: sh.name, intType: sh.intType, table: table, id: uint32(i), payload: payload, + columns: []string{"id", "v"}, + line: rowFilterStoredLine(t, table, uint32(i)), + }) + } + require.True(t, anyStored, "corpus for %s must contain insertable payloads", sh.name) + } + p := policy.Policy{Tables: tables} + withPolicy(t, p) + + // The hub's evaluator over the app's type layer, which createTable's + // refreshes already bound. + require.NotNil(t, env(t).types, "the api role wires a type layer") + eval := stream.NewRowEvaluator(env(t).types) + + byShape := map[string][]string{} + for _, sh := range shapes { + byShape[sh.name] = sh.constants + } + cells, mathCells, admitted := 0, 0, 0 + for _, r := range rows { + stored := rowFilterStoredInt(t, r.intType, r.line) + for i, constant := range byShape[r.shape] { + for _, op := range diffOps { + cells++ + role := cellRole(i, op) + got := rowFilterStreamVerdict(t, eval, &p, role, r.table, r.columns, r.line, nil) + want, sqlErr := rowFilterQueryVerdict(t, r.table, r.id, role, nil) + if want { + admitted++ + } + if got != want { + t.Errorf("%s: stored %v %s %q — stream says %v, /v1/query says %v (query err: %v)", + r.shape, r.payload, op.sql, constant, got, want, sqlErr) + } + if exact, ok := intExpected(r.intType, stored, op.sql, constant); ok { + mathCells++ + if want != exact || got != exact { + t.Errorf("%s: stored %v %s %q — admitted stream=%v query=%v, the mathematically correct answer is %v", + r.shape, r.payload, op.sql, constant, got, want, exact) + } + } + } + } + + // A multi-element _in from a claim array: each element is cast on its + // own, so the elements that fit decide and the rest drop out — 2^64+5 + // must not wrap onto a row holding 5. + if r.intType != "" { + for _, set := range intInSets { + cells++ + mathCells++ + claims := map[string]any{"ids": toAnys(set)} + got := rowFilterStreamVerdict(t, eval, &p, claimInRole, r.table, r.columns, r.line, claims) + want, sqlErr := rowFilterQueryVerdict(t, r.table, r.id, claimInRole, claims) + exact := false + for _, c := range set { + if eq, ok := intExpected(r.intType, stored, "=", c); ok && eq { + exact = true + } + } + if got != want || want != exact { + t.Errorf("%s: stored %v IN %v — stream says %v, /v1/query says %v (query err: %v), correct is %v", + r.shape, r.payload, set, got, want, sqlErr, exact) + } + } + } + } + // A harness that answered "no row" everywhere would agree with itself. + require.Positive(t, admitted, "some cell must admit its row on the query path") + t.Logf("%d cells compared stream against /v1/query (%d admitted), %d of them also against the exact answer", cells, admitted, mathCells) +} + +// diffOp is one operator under the differential: its SQL spelling, for the +// oracle and the messages, and the role-name spelling. +type diffOp struct{ sql, name string } + +var diffOps = []diffOp{{"=", "eq"}, {"!=", "neq"}, {">", "gt"}, {"<", "lt"}, {"in", "in"}} + +// claimInRole is the role whose _in reads the ids claim. +const claimInRole = "claim_in" + +// cellRole names the role of one (constant, operator) cell. Role names repeat +// across tables: each table's grant is its own. +func cellRole(constant int, op diffOp) string { + return fmt.Sprintf("c%d_%s", constant, op.name) +} + +// rowFilterGrant reads a table, filtered by one operator on v with a literal +// constant. +func rowFilterGrant(t *testing.T, op diffOp, constant string) policy.RolePermissions { + t.Helper() + f := policy.Filter{} + switch op.name { + case "eq": + f.Eq = &constant + case "neq": + f.Neq = &constant + case "gt": + f.Gt = &constant + case "lt": + f.Lt = &constant + case "in": + // A placeholder-free _in template resolves to a one-element set, so this + // is the IN (…) renderer on both surfaces with one bound value. + f.In = &constant + default: + t.Fatalf("unknown op %q", op.name) + } + return policy.RolePermissions{Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"v": f}}} +} + +// rowFilterStreamVerdict resolves role's grant through the full production +// path (Evaluate → Prepare → Visible) and reports whether the stream would +// deliver this stored row. +func rowFilterStreamVerdict(t *testing.T, eval stream.RowEvaluator, p *policy.Policy, role, table string, columns []string, line []byte, claims map[string]any) bool { + t.Helper() + perms := policy.Evaluate(p, role, table, "select", claims) + require.True(t, perms.Allowed) + + view, err := eval.Prepare(tenant.Default, table, columns, line) + if err != nil { + return false // withheld: no view, no row + } + defer view.Close() + visible, _ := view.Visible(perms) + return visible +} + +// rowFilterQueryVerdict asks the production /v1/query, as role with claims, +// whether it returns the stored row. A query ClickHouse rejects means the role +// reads no rows on that path; any other failure is the harness's and fails the +// test, so a broken token or policy cannot pass as "withheld on both". +func rowFilterQueryVerdict(t *testing.T, table string, id uint32, role string, claims map[string]any) (bool, error) { + t.Helper() + body, err := json.Marshal(map[string]any{ + "columns": []string{"id"}, + "filters": []any{map[string]any{"column": "id", "op": "eq", "value": id}}, + }) + require.NoError(t, err) + got := postJSONAs(t, env(t).baseURL+"/v1/query?table="+url.QueryEscape(table), string(body), bearer(t, role, claims)) + switch { + case got.status == http.StatusOK: + case got.status == http.StatusBadRequest && got.Code == "clickhouse.rejected": + return false, fmt.Errorf("HTTP %d: %s", got.status, got.raw) + default: + t.Fatalf("role %s on %s: HTTP %d: %s", role, table, got.status, got.raw) + } + var out []map[string]any + require.NoError(t, json.Unmarshal([]byte(got.raw), &out)) + require.LessOrEqual(t, len(out), 1, "id is unique per table") + return len(out) == 1, nil +} + +// Boundary constants for the integer shapes: each width's own edges, the +// values that wrapped under the plain String binding, and spellings that are +// not canonical. +var intBoundaryConstants = func() []string { + p := func(n uint) *big.Int { return new(big.Int).Lsh(big.NewInt(1), n) } + add := func(a *big.Int, d int64) string { return new(big.Int).Add(a, big.NewInt(d)).String() } + neg := func(a *big.Int) *big.Int { return new(big.Int).Neg(a) } + return []string{ + "0", "1", "5", "-1", "-5", "255", "256", "4294967295", "4294967296", + add(p(63), 0), add(p(63), -1), add(neg(p(63)), -1), add(neg(p(63)), 0), + add(p(64), 0), add(p(64), -1), add(p(64), 5), + add(p(127), 0), add(p(127), -1), add(neg(p(127)), -1), add(neg(p(127)), 0), + add(p(128), 0), add(p(128), -1), + add(p(255), 0), add(p(255), -1), add(neg(p(255)), -1), add(neg(p(255)), 0), + add(p(256), 0), add(p(256), -1), add(p(256), 5), + "007", "+5", "5.0", "1e3", "abc", "", + } +}() + +// intInSets are claim arrays for the multi-element _in cases. +var intInSets = [][]string{ + {"18446744073709551621", "0"}, // 2^64+5 must not wrap onto 5 + {"5", "007", "abc"}, // the junk elements drop out, 5 still decides + {"-1", "340282366920938463463374607431768211456", "1"}, // 2^128 + {"115792089237316195423570985008687907853269984665640564039457584007913129639941", "+5", "5.0"}, // 2^256+5 +} + +// diffShape is one column type under the differential. intType names the +// bare integer type the column holds, "" for a non-integer column; it is +// declared here rather than derived with chsql.IntegerType, so a regression in +// that derivation shows up as a wrong answer instead of a skipped oracle. +type diffShape struct { + name string + ddl string + intType string + payloads []any + constants []string +} + +// intShape is a differential shape for one integer type: a row at each edge of +// its domain (and a NULL for a Nullable column), filtered by every boundary +// constant. +func intShape(name, ddl, intType string) diffShape { + lo, hi := intDomain(intType) + payloads := []any{"0", "1", "5", hi.String()} + if lo.Sign() < 0 { + payloads = append(payloads, lo.String(), "-5") + } + if strings.HasPrefix(ddl, "Nullable(") { + payloads = append(payloads, nil) + } + return diffShape{name: name, ddl: ddl, intType: intType, payloads: payloads, constants: intBoundaryConstants} +} + +// intDomain is an integer type's [min, max]. +func intDomain(name string) (*big.Int, *big.Int) { + bits, err := strconv.Atoi(strings.TrimPrefix(strings.TrimPrefix(name, "U"), "Int")) + if err != nil { + panic(name) + } + one := big.NewInt(1) + if strings.HasPrefix(name, "U") { + return big.NewInt(0), new(big.Int).Sub(new(big.Int).Lsh(one, uint(bits)), one) + } + half := new(big.Int).Lsh(one, uint(bits-1)) + return new(big.Int).Neg(half), new(big.Int).Sub(half, one) +} + +// rowFilterStoredInt reads the stored value back off the published line for an +// integer column; nil for NULL or a non-integer column. +func rowFilterStoredInt(t *testing.T, intType string, line []byte) *big.Int { + t.Helper() + if intType == "" { + return nil + } + dec := json.NewDecoder(bytes.NewReader(line)) + dec.UseNumber() + var cols []any + require.NoError(t, dec.Decode(&cols)) + require.Len(t, cols, 2) + var text string + switch v := cols[1].(type) { + case nil: + return nil + case json.Number: + text = v.String() + case string: + text = v + default: + t.Fatalf("stored integer came back as %T", v) + } + n, ok := new(big.Int).SetString(text, 10) + require.True(t, ok, "stored integer %q", text) + return n +} + +// intExpected is the mathematically correct verdict for an integer column: a +// constant that is not the canonical spelling of a value the column can hold +// admits nothing, and a NULL row is never admitted. ok is false for a +// non-integer column, which has no such oracle here. +func intExpected(intType string, stored *big.Int, op, constant string) (bool, bool) { + if intType == "" { + return false, false + } + lo, hi := intDomain(intType) + v, ok := new(big.Int).SetString(constant, 10) + if !ok || v.String() != constant || v.Cmp(lo) < 0 || v.Cmp(hi) > 0 || stored == nil { + return false, true + } + cmp := stored.Cmp(v) + switch op { + case "=", "in": + return cmp == 0, true + case "!=": + return cmp != 0, true + case "<": + return cmp < 0, true + case ">": + return cmp > 0, true + } + panic(op) +} + +func toAnys(ss []string) []any { + out := make([]any, len(ss)) + for i, s := range ss { + out[i] = s + } + return out +} + +// rowFilterStoredLine reads one stored row back as the positional +// JSONCompactEachRow line the ingest path publishes for it — the exact bytes the +// stream evaluates, produced by the server rather than reconstructed here. +func rowFilterStoredLine(t *testing.T, table string, id uint32) []byte { + t.Helper() + q := url.Values{} + q.Set("database", testCHDatabase) + q.Set("param_target_table", table) + q.Set("param_id", fmt.Sprint(id)) + q.Set("query", "SELECT * FROM {target_table:Identifier} WHERE id = {id:UInt32} FORMAT JSONCompactEachRow") + body := rowFilterCH(t, http.MethodGet, q, nil) + line := strings.TrimRight(string(body), "\n") + require.NotEmpty(t, line, "stored row must come back") + require.NotContains(t, line, "\n", "exactly one row per id") + return []byte(line) +} + +// rowFilterInsert inserts one JSONEachRow row over ClickHouse HTTP with the same +// parsing settings the ingest worker pins, and returns ClickHouse's verdict as +// an error (nil on 2xx). +func rowFilterInsert(t *testing.T, table string, row map[string]any) error { + t.Helper() + body, err := json.Marshal(row) + require.NoError(t, err) + + q := url.Values{} + q.Set("database", testCHDatabase) + q.Set("param_target_table", table) + q.Set("query", "INSERT INTO {target_table:Identifier} FORMAT JSONEachRow") + for k, v := range typelayer.InsertSettings() { + q.Set(k, v) + } + req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, + env(t).chHTTPURL+"?"+q.Encode(), bytes.NewReader(body)) + require.NoError(t, err) + req.Header.Set("Content-Type", "application/json") + req.Header.Set("X-ClickHouse-User", testCHUser) + req.Header.Set("X-ClickHouse-Key", testCHPassword) + + resp, err := http.DefaultClient.Do(req) + require.NoError(t, err) + defer func() { _ = resp.Body.Close() }() + if resp.StatusCode >= 300 { + msg, _ := io.ReadAll(io.LimitReader(resp.Body, 512)) + return fmt.Errorf("HTTP %d: %s", resp.StatusCode, msg) + } + _, _ = io.Copy(io.Discard, resp.Body) + return nil +} + +// rowFilterCH runs one ClickHouse HTTP request and returns the body, failing the +// test on anything but 2xx. +func rowFilterCH(t *testing.T, method string, q url.Values, body io.Reader) []byte { + t.Helper() + req, err := http.NewRequestWithContext(context.Background(), method, env(t).chHTTPURL+"?"+q.Encode(), body) + require.NoError(t, err) + req.Header.Set("X-ClickHouse-User", testCHUser) + req.Header.Set("X-ClickHouse-Key", testCHPassword) + resp, err := http.DefaultClient.Do(req) + require.NoError(t, err) + defer func() { _ = resp.Body.Close() }() + out, err := io.ReadAll(resp.Body) + require.NoError(t, err) + require.Less(t, resp.StatusCode, 300, "clickhouse: %s", out) + return out +} diff --git a/tests/integration/setup_test.go b/tests/integration/setup_test.go index 48f5b52e..dadef58b 100644 --- a/tests/integration/setup_test.go +++ b/tests/integration/setup_test.go @@ -16,10 +16,12 @@ import ( "errors" "fmt" "io" + "maps" "net" "net/http" "os" "path/filepath" + "slices" "strconv" "strings" "sync/atomic" @@ -36,13 +38,26 @@ import ( "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/mq/natstest" + "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/testutil" + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) const ( testCHPassword = "test" testCHDatabase = "default" testCHUser = "default" + // testOperatorKey is the shared app's operator key: the settings reload + // route takes it whatever policy is adopted. + testOperatorKey = "it-shared-operator-key" +) + +// The suite's default roles.json and policies.json: default_role is the admin +// role, so a plain unauthenticated request runs as a privileged caller. +var ( + defaultRoles = []byte(`{"roles": ["admin"]}`) + defaultPolicies = []byte(`{"default_role": "admin"}`) ) // testEnv holds the shared infrastructure available to every test. @@ -53,6 +68,12 @@ type testEnv struct { embeddedMQ mq.Broker baseURL string // the wired API server, e.g. http://127.0.0.1:41234 registry *discovery.SchemaRegistry + // types is the wired app's type layer, bound from the default tenant's + // discovery by the app's own refresh hook. + types *typelayer.Engine + // settingsDir is the shared app's settings directory, which withPolicy + // rewrites and reloads. + settingsDir string // dynamoEndpoint is dynamodb-local, for the DynamoDB dedupe backend's // tests; the wired app does not use it. dynamoEndpoint string @@ -177,13 +198,19 @@ func setup() (int, func()) { DataDir: dataDir, Server: config.Server{ShutdownTimeout: 10}, ClickHouse: config.ClickHouse{Password: testCHPassword}, - MQ: config.MQ{Backend: config.MQEmbedded}, - Cache: config.Cache{Backend: config.CacheLocal, L1MaxCost: 1 << 30}, // 1 GB - Dedupe: config.Dedupe{Backend: config.DedupePebble}, - Coord: config.Coord{Backend: config.CoordLocal}, - Roles: config.AllRoles(), - Settings: config.Settings{Dir: settingsDir}, - } + // The JWT secret the suite mints tokens with (bearer), for the tests + // that run as a restricted role under a policy of their own. + Auth: config.Auth{JWTSecret: testutil.TestJWTSecret, OperatorKey: testOperatorKey}, + MQ: config.MQ{Backend: config.MQEmbedded}, + Cache: config.Cache{Backend: config.CacheLocal, L1MaxCost: 1 << 30}, // 1 GB + Dedupe: config.Dedupe{Backend: config.DedupePebble}, + Coord: config.Coord{Backend: config.CoordLocal}, + Roles: config.AllRoles(), + Settings: config.Settings{Dir: settingsDir}, + } + // The api role opens the type layer, which needs the chtypes artifact for + // the container's ClickHouse line (scripts/fetch-chtypes.sh): without it + // app.New refuses, naming where it looked. a, err := app.New(ctx, app.Options{Config: cfg, Listener: ln}) if err != nil { _ = ln.Close() @@ -219,6 +246,9 @@ func setup() (int, func()) { embeddedMQ: a.MQ(), baseURL: baseURL, registry: a.Registry(), + types: a.Types(), + + settingsDir: settingsDir, dynamoEndpoint: endpoint, } @@ -229,9 +259,10 @@ func setup() (int, func()) { // pointed at the testcontainer and a dev-style policy: default_role is the // admin role, so the suite's plain unauthenticated requests exercise // functionality as a privileged caller and can hit admin-gated endpoints -// without minting JWTs. Auth enforcement is covered by the internal/auth -// unit tests and the e2e SDK suite. The stream budget is shrunk to 1 GiB -// like the e2e fixture so the scratch directory stays small. +// without minting JWTs. A test that needs a restricted role adopts a policy +// of its own (withPolicy) and sends a token for the role (bearer). The stream +// budget is shrunk to 1 GiB like the e2e fixture so the scratch directory +// stays small. func writeTestSettings(ch *chInstance) (string, error) { files, err := tenantSettings(ch, testCHDatabase) if err != nil { @@ -272,11 +303,92 @@ func tenantSettings(ch *chInstance, database string) (map[string][]byte, error) if files[settings.FileConfig], err = json.MarshalIndent(doc, "", " "); err != nil { return nil, err } - files[settings.FileRoles] = []byte(`{"roles": ["admin"]}`) - files[settings.FilePolicies] = []byte(`{"default_role": "admin"}`) + files[settings.FileRoles] = defaultRoles + files[settings.FilePolicies] = defaultPolicies return files, nil } +// withPolicy adopts p as the shared app's access-control policy for the +// calling test, the way an operator changes one: policies.json rewritten, +// with roles.json declaring every role it grants, then a reload through the +// ops route. The suite's default policy is restored when the test ends. +// default_role and admin_role stay the admin role, so unauthenticated +// requests keep running as a privileged caller and a restricted role is +// reached with a token for it (bearer). Not for parallel tests: there is one +// shared policy. +func withPolicy(t *testing.T, p policy.Policy) { + t.Helper() + p.DefaultRole, p.AdminRole = "admin", "admin" + roles := map[string]bool{"admin": true} + for _, grants := range p.Tables { + for role := range grants { + roles[role] = true + } + } + rolesDoc, err := json.Marshal(settings.RolesFile{Roles: slices.Sorted(maps.Keys(roles))}) + if err != nil { + t.Fatalf("roles.json: %v", err) + } + policyDoc, err := json.Marshal(p) + if err != nil { + t.Fatalf("policies.json: %v", err) + } + // The roles go in before the policy granting them and come out after it, + // so the directory watcher, which may reload between the two files, only + // ever sees a valid pair. + adoptSettings(t, settingsFile{settings.FileRoles, rolesDoc}, settingsFile{settings.FilePolicies, policyDoc}) + t.Cleanup(func() { + adoptSettings(t, settingsFile{settings.FilePolicies, defaultPolicies}, settingsFile{settings.FileRoles, defaultRoles}) + }) +} + +// settingsFile is one file of the settings directory and its new content. +type settingsFile struct { + name string + data []byte +} + +// adoptSettings writes files, in order, into the shared app's settings +// directory and reloads it, failing the test unless the reload adopts them. +// Each file is renamed into place, so the directory watcher never reads one +// half-written. +func adoptSettings(t *testing.T, files ...settingsFile) { + t.Helper() + e := env(t) + for _, f := range files { + tmp := filepath.Join(filepath.Dir(e.settingsDir), filepath.Base(e.settingsDir)+"."+f.name+".tmp") + if err := os.WriteFile(tmp, f.data, 0o600); err != nil { + t.Fatalf("write %s: %v", f.name, err) + } + if err := os.Rename(tmp, filepath.Join(e.settingsDir, f.name)); err != nil { + t.Fatalf("install %s: %v", f.name, err) + } + } + req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, e.baseURL+"/v1/ops/settings/reload", nil) + if err != nil { + t.Fatalf("reload request: %v", err) + } + req.Header.Set("X-Operator-Key", testOperatorKey) + resp, err := http.DefaultClient.Do(req) + if err != nil { + t.Fatalf("reload: %v", err) + } + defer func() { _ = resp.Body.Close() }() + body, _ := io.ReadAll(resp.Body) + if resp.StatusCode != http.StatusOK { + t.Fatalf("settings reload not adopted: %d %s", resp.StatusCode, body) + } +} + +// bearer is an Authorization header value carrying a token for role, signed +// with the shared app's JWT secret, with claims beside the role claim. +func bearer(t *testing.T, role string, claims map[string]any) string { + t.Helper() + all := map[string]any{"role": role} + maps.Copy(all, claims) + return "Bearer " + testutil.MakeJWT(t, all) +} + // writeSettingsFiles writes one tenant's files into dir. func writeSettingsFiles(dir string, files map[string][]byte) error { if err := os.MkdirAll(dir, 0o750); err != nil { @@ -350,11 +462,11 @@ func (c *chInstance) httpURL() string { return fmt.Sprintf("http://%s:%s", c. // race; the dominant flake mode tracked in #70. func startClickHouse(ctx context.Context) (*chInstance, error) { chReq := testcontainers.ContainerRequest{ - // Pinned: 26.8 reads bare numbers in DateTime64 columns as epoch seconds, - // not ticks at column precision — CanonicalizeTimestamps still models the - // pre-26.8 rule (TestTimestampCanonicalization_DifferentialAgainstClickHouse - // catches the divergence). Bump the pin together with the canonicalizer (#536). - Image: "clickhouse/clickhouse-server:26.6.3.62", + // Pinned to a line chtypes.lock has an artifact for: the type layer + // answers with the artifact matching the server's own version, so + // bumping the line means locking that line's artifact too (#536). + // deployments/compose pins the same image. + Image: "clickhouse/clickhouse-server:26.8.15.10", ExposedPorts: []string{"9000/tcp", "8123/tcp"}, Env: map[string]string{"CLICKHOUSE_PASSWORD": testCHPassword}, WaitingFor: wait.ForAll( diff --git a/tests/integration/testdata/query_types_pin.json b/tests/integration/testdata/query_types_pin.json new file mode 100644 index 00000000..58f2f03f --- /dev/null +++ b/tests/integration/testdata/query_types_pin.json @@ -0,0 +1 @@ +[{"i8":-8,"i16":-16,"i32":-32,"i64":-9007199254740993,"u8":8,"u16":16,"u32":32,"u64":18446744073709551615,"f32":0.1,"f64":0.1,"dec":12.5,"dec64":1.5,"s":"hello","ls":"lc","fs":"ab\u0000\u0000","uu":"11111111-2222-3333-4444-555555555555","en":"a","bl":true,"ip4":"10.0.0.1","ip6":"::1","d":"2026-01-15","dt":"2026-01-15 10:30:00","dt64":"2026-01-15 10:30:00.123","arr":["a","b"],"m":{"k":7},"nn":42,"nnull":null}] diff --git a/tests/integration/timestamp_canonicalization_test.go b/tests/integration/timestamp_canonicalization_test.go deleted file mode 100644 index c0463ed7..00000000 --- a/tests/integration/timestamp_canonicalization_test.go +++ /dev/null @@ -1,163 +0,0 @@ -//go:build integration - -package tests - -import ( - "bytes" - "context" - "encoding/json" - "fmt" - "io" - "net/http" - "net/url" - "testing" - "time" - - "github.com/stretchr/testify/require" - - "github.com/Wave-RF/WaveHouse/internal/discovery" -) - -// TestTimestampCanonicalization_DifferentialAgainstClickHouse pins the #372 -// invariant with ClickHouse itself as the oracle (PR #402 review): for every -// corpus input × timestamp column shape, inserting the raw producer value and -// inserting CanonicalizeTimestamps' output must both succeed or both fail, and -// when both succeed store the same instant — anything else means the rewrite -// changed what ClickHouse stores or accepts. Inserts go through the worker's -// exact HTTP surface (JSONEachRow, date_time_input_format=best_effort), so the -// oracle is the production parse, not a lookalike. -func TestTimestampCanonicalization_DifferentialAgainstClickHouse(t *testing.T) { - colTypes := []struct { - name string - ddl string - }{ - {"datetime", "DateTime"}, - {"datetime_nyc", "DateTime('America/New_York')"}, - {"datetime64_3", "DateTime64(3)"}, - {"datetime64_6_tokyo", "DateTime64(6, 'Asia/Tokyo')"}, - // Precision 9 has its own ceiling (Int64 nanoseconds end 2262-04-11, past - // which an insert fails outright instead of saturating), and precision 0 - // pins the DateTime64-kind rules on a second-granular column. - {"datetime64_9", "DateTime64(9)"}, - {"datetime64_0_utc", "DateTime64(0, 'UTC')"}, - } - - // One entry per spelling family, each found or pinned by differentially - // fuzzing the canonicalizer against a live ClickHouse (PR #402 review). The - // digit runs, number shapes, out-of-range instants, and garbage document the - // pass-through side: they must reach ClickHouse verbatim and get its - // verdict, never a rewrite. - corpus := []any{ - "2026-06-21T04:00:00Z", // canonical already - "2026-06-21T06:30:00+02:30", // offset form - "2026-06-21 04:00:00", // zone-less space form - "2026-06-21T04:00:00", // zone-less T form - "2026-06-21", // date-only - "2026-06-21 04:00:00.123456", // zone-less with fraction - "2026-06-21T04:00:00.9999Z", // fraction beyond column precision - "2026-06-21T04:00:00,999Z", // comma fraction: ISO 8601 yes, ClickHouse no - "1750478400", // 10-digit Unix string - "999999999", // 9-digit Unix string - "1750478400.5", // fractional Unix string: DateTime64-only to ClickHouse - "1750478400.123456789", // ns-exact fraction (float64 would corrupt it) - "1750478400.9999999995", // >9 fraction digits: ClickHouse truncates, never rounds - float64(1750478400), // integer number: seconds to DateTime, *ticks* to DateTime64 - float64(1750478400.5), // non-integer number: ClickHouse rejects for every kind - float64(1750478400500), // epoch-ms number: DateTime64(3)'s natural ticks shape - json.Number("1750478400"), // integer seconds as the production-decoded type - json.Number("1750478400.5"), // non-integer json.Number: pass-through, ClickHouse rejects - json.Number("1750478400123456789"), // 19-digit ns epoch > 2^53: rewritten only on DateTime64(9) - "20260711", // 8 digits: YYYYMMDD to ClickHouse - "20260711150000", // 14 digits: YYYYMMDDhhmmss - "202607111500", // 12 digits: ClickHouse rejects - "1752278400000", // 13 digits: epoch milliseconds to ClickHouse - "1750478400123456", // 16 digits: epoch microseconds to ClickHouse - "2026", // 4 digits: a year to ClickHouse - "1e9", // not a timestamp - "-100", // not a timestamp - "banana", // garbage - "2026-11-01 01:30:00", // DST fall-back: ambiguous local time - "2026-03-08 02:30:00", // DST spring-forward: nonexistent local time - "1960-01-01T00:00:00Z", // pre-range: ClickHouse saturates, spelling-dependently - "2300-06-30 12:30:00", // beyond DateTime64's ceiling, zone-less - } - - for _, ct := range colTypes { - t.Run(ct.name, func(t *testing.T) { - // Independent tables per shape; parallel keeps the six shapes from - // serializing ~380 single-row inserts against the suite timeout. - t.Parallel() - rawTable := createTable(t, "id UInt32, v "+ct.ddl, "ORDER BY id") - canonTable := createTable(t, "id UInt32, v "+ct.ddl, "ORDER BY id") - schema := env(t).registry.Get(rawTable) - require.NotNil(t, schema, "registry must discover the raw table") - - for i, input := range corpus { - id := uint32(i) - rawErr := insertBestEffort(t, rawTable, map[string]any{"id": id, "v": input}) - - canonData := map[string]any{"id": id, "v": input} - discovery.CanonicalizeTimestamps(schema, canonData) - canonErr := insertBestEffort(t, canonTable, canonData) - - if (rawErr == nil) != (canonErr == nil) { - t.Errorf("input %v: asymmetric insertability — raw err=%v, canonicalized (%v) err=%v", - input, rawErr, canonData["v"], canonErr) - continue - } - if rawErr != nil { - continue // both rejected — consistent - } - rawStored := selectInstant(t, rawTable, id) - canonStored := selectInstant(t, canonTable, id) - if !rawStored.Equal(canonStored) { - t.Errorf("input %v: stored instants differ — raw %s vs canonicalized (%v) %s", - input, rawStored.UTC().Format(time.RFC3339Nano), - canonData["v"], canonStored.UTC().Format(time.RFC3339Nano)) - } - } - }) - } -} - -// insertBestEffort inserts one JSONEachRow row over ClickHouse HTTP with -// date_time_input_format=best_effort — the exact settings the ingest worker -// uses (see IngestWorker.insertToClickHouse) — and returns ClickHouse's -// verdict as an error (nil on 2xx). -func insertBestEffort(t *testing.T, table string, row map[string]any) error { - t.Helper() - body, err := json.Marshal(row) - require.NoError(t, err) - - q := url.Values{} - q.Set("database", testCHDatabase) - q.Set("param_target_table", table) - q.Set("query", "INSERT INTO {target_table:Identifier} FORMAT JSONEachRow") - q.Set("date_time_input_format", "best_effort") - - req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, - env(t).chHTTPURL+"?"+q.Encode(), bytes.NewReader(body)) - require.NoError(t, err) - req.Header.Set("Content-Type", "application/json") - req.Header.Set("X-ClickHouse-User", testCHUser) - req.Header.Set("X-ClickHouse-Key", testCHPassword) - - resp, err := http.DefaultClient.Do(req) - require.NoError(t, err) - defer func() { _ = resp.Body.Close() }() - if resp.StatusCode >= 300 { - msg, _ := io.ReadAll(io.LimitReader(resp.Body, 512)) - return fmt.Errorf("HTTP %d: %s", resp.StatusCode, msg) - } - _, _ = io.Copy(io.Discard, resp.Body) - return nil -} - -// selectInstant reads back the single stored timestamp as a time.Time instant. -func selectInstant(t *testing.T, table string, id uint32) time.Time { - t.Helper() - var v time.Time - require.NoError(t, env(t).chConn.QueryRow(context.Background(), - fmt.Sprintf("SELECT v FROM %s WHERE id = ?", table), id).Scan(&v)) - return v -} diff --git a/tests/integration/typelayer_wire_test.go b/tests/integration/typelayer_wire_test.go new file mode 100644 index 00000000..e95a2830 --- /dev/null +++ b/tests/integration/typelayer_wire_test.go @@ -0,0 +1,171 @@ +//go:build integration + +package tests + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "net/url" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// TestTypelayerWire_PublishedRowIsTheStoredRow is the one claim the whole +// migration rests on: the bytes WaveHouse publishes — to NATS, to the SSE +// stream, to the DLQ — are the row ClickHouse itself will hold. They are +// produced by the server's own writer at validation time, so a subscriber +// reading the stream and a client reading /v1/query cannot disagree about a +// value, and the worker's INSERT re-parses bytes the server already wrote. +// +// It replaces the timestamp differential oracle (a corpus of producer +// spellings × column shapes, asserting a Go canonicalizer landed on the same +// instant as the raw value). That oracle existed to prove a hand-written +// parser matched ClickHouse; there is no longer a second parser to disagree, +// so one end-to-end identity check is the whole remaining claim. Deliberately +// one test, not a corpus. +func TestTypelayerWire_PublishedRowIsTheStoredRow(t *testing.T) { + e := env(t) + ctx := context.Background() + before := time.Now().UTC().Add(-time.Minute) + + table := createTable(t, + "user_id String, "+ + "ts DateTime('UTC'), "+ + "ts_ms DateTime64(3, 'UTC'), "+ + "amount Decimal(10, 2), "+ + "small UInt8, "+ + "tags Array(String), "+ + "maybe Nullable(String)", + "ORDER BY user_id", + ) + + // Every value in a spelling ClickHouse does NOT store it in: an offset + // timestamp, epoch milliseconds for a DateTime64(3), a string decimal, and + // an integer past the column's width (which wraps — the stored truth, not + // the producer's). The milliseconds are a quoted string, which 26.6 and + // 26.8 both read as ticks at the column's precision; a bare JSON number is + // ticks on 26.6 but epoch seconds on 26.8 (clamped to 9999-12-31 at this + // magnitude), which would pin the server line rather than the wiring. + body := `{"user_id":"erin",` + + `"ts":"2026-06-21T06:00:00+02:00",` + + `"ts_ms":"1782014400123",` + + `"amount":"12.50",` + + `"small":256,` + + `"tags":["a","b"],` + + `"maybe":null}` + + resp, err := http.Post( + e.baseURL+"/v1/ingest?table="+url.QueryEscape(table), + "application/json", + strings.NewReader(body), + ) + require.NoError(t, err) + defer resp.Body.Close() + require.Equal(t, http.StatusOK, resp.StatusCode) + + // 30s upper bound for the worker's 5s batch window plus loaded-runner slack. + require.Eventually(t, func() bool { + var count uint64 + err := e.chConn.QueryRow(ctx, + fmt.Sprintf("SELECT count() FROM %s WHERE user_id = 'erin'", table), + ).Scan(&count) + return err == nil && count == 1 + }, 30*time.Second, 250*time.Millisecond, "row never landed") + + // Read the stored row back in the same format the wire carries, so the two + // are comparable as bytes rather than through two different renderings. + stored := selectJSONCompactRow(t, e.chHTTPURL, + fmt.Sprintf("SELECT user_id, ts, ts_ms, amount, small, tags, maybe FROM %s WHERE user_id = 'erin'", table)) + + wire := publishedWireRow(t, e.baseURL, table, before) + assert.Equal(t, stored, wire, + "the published row and the stored row must be the same values: wire=%v stored=%v", wire, stored) + + // And spot-check that these really are the coerced values, not the + // producer's spellings passed through. + require.Len(t, wire, 7) + assert.Equal(t, "2026-06-21 04:00:00", wire[1], "offset applied, rendered in the column's zone") + assert.Equal(t, "2026-06-21 04:00:00.123", wire[2], "ticks at the column's precision") + assert.EqualValues(t, 0, wire[4], "256 into a UInt8 wraps — the stored truth") +} + +// publishedWireRow replays the table's SSE stream from before the insert, so +// the test reads the published bytes without racing the publish. +func publishedWireRow(t *testing.T, serverURL, table string, since time.Time) []any { + t.Helper() + req, err := http.NewRequestWithContext(context.Background(), http.MethodGet, + serverURL+"/v1/stream?table="+url.QueryEscape(table)+ + "&since="+url.QueryEscape(since.Format(time.RFC3339Nano)), nil) + require.NoError(t, err) + req.Header.Set("Accept", "text/event-stream") + + client := &http.Client{Timeout: 15 * time.Second} + resp, err := client.Do(req) + require.NoError(t, err) + defer resp.Body.Close() + require.Equal(t, http.StatusOK, resp.StatusCode) + + cells, ok := firstDataRow(t, resp.Body) + require.True(t, ok, "no data frame arrived on the replay stream") + return cells +} + +// firstDataRow reads SSE frames until one carries a row, and returns its cells. +func firstDataRow(t *testing.T, body interface{ Read([]byte) (int, error) }) ([]any, bool) { + t.Helper() + buf := make([]byte, 0, 8192) + chunk := make([]byte, 4096) + deadline := time.Now().Add(15 * time.Second) + for time.Now().Before(deadline) { + n, err := body.Read(chunk) + if n > 0 { + buf = append(buf, chunk[:n]...) + for _, line := range strings.Split(string(buf), "\n") { + payload, found := strings.CutPrefix(line, "data: ") + if !found { + continue + } + var frame struct { + Row []any `json:"row"` + } + if json.Unmarshal([]byte(payload), &frame) == nil && frame.Row != nil { + return frame.Row, true + } + } + } + if err != nil { + return nil, false + } + } + return nil, false +} + +// selectJSONCompactRow reads one row back through ClickHouse's HTTP interface in +// JSONCompactEachRow — the same writer that produced the published row, so the +// two renderings are comparable without a second interpretation step. +func selectJSONCompactRow(t *testing.T, chHTTPURL, query string) []any { + t.Helper() + q := url.Values{} + q.Set("database", testCHDatabase) + q.Set("query", query+" FORMAT JSONCompactEachRow") + + req, err := http.NewRequestWithContext(context.Background(), http.MethodGet, chHTTPURL+"?"+q.Encode(), nil) + require.NoError(t, err) + req.Header.Set("X-ClickHouse-User", testCHUser) + req.Header.Set("X-ClickHouse-Key", testCHPassword) + + resp, err := (&http.Client{Timeout: 15 * time.Second}).Do(req) + require.NoError(t, err) + defer resp.Body.Close() + require.Equal(t, http.StatusOK, resp.StatusCode) + + var cells []any + require.NoError(t, json.NewDecoder(resp.Body).Decode(&cells)) + return cells +} From eb5ba5baa32573d679154a12f603aca66fd9909c Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:20:53 -0400 Subject: [PATCH 18/70] test(ingest): feed DateTime64 the way ClickHouse 26.8 reads a number 26.8 reads a bare JSON number in a DateTime64 column as epoch seconds, so the epoch-milliseconds value the test sent now clamps to the type's maximum. Send fractional seconds instead. The CI docs and the setup-env comment name the 26.8 line and SDK v0.5.1. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- .github/actions/setup-env/action.yml | 2 +- .github/workflows/README.md | 6 +++--- internal/api/ingest_test.go | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/actions/setup-env/action.yml b/.github/actions/setup-env/action.yml index a4cea5ee..462721c6 100644 --- a/.github/actions/setup-env/action.yml +++ b/.github/actions/setup-env/action.yml @@ -46,7 +46,7 @@ # pipeline. See #132. # 7. chtypes artifact cache (~/.cache/chtypes/artifacts/abi6) keyed on # chtypes.lock — the pinned ClickHouse-version .so/.dylib the SDK -# dlopens. Measured for today's one pinned line (26.6, linux-amd64): +# dlopens. Measured for the 26.6 line (linux-amd64), pinned before 26.8: # 316 MiB of libchtypes.so on disk, ~65 MiB as a stored archive, and # 85 MiB over the wire on a MISS (the upstream .tar.gz). Fetched via # scripts/fetch-chtypes.sh (--frozen: refuses anything the lock diff --git a/.github/workflows/README.md b/.github/workflows/README.md index 66348434..2e9633d2 100644 --- a/.github/workflows/README.md +++ b/.github/workflows/README.md @@ -80,7 +80,7 @@ Queue settings live in the `main branch protection` ruleset's `merge_queue` rule | pnpm store | `pnpm--` | any node job on miss | Store path resolved from pnpm at runtime. docs-build prunes before its save on a key rotation. | | Playwright Chromium | `playwright--` | docs-build | rehype-mermaid renders via headless Chrome at docs build. | | Astro content collections | `astro--` | lint / docs-build | Warm `astro check`/`build` skip unchanged content. | -| chtypes artifact | `chtypes-abi6---` | unit / integration / e2e via `setup-env` (shared) | `~/.cache/chtypes/artifacts/abi6` — the pinned ClickHouse-version `.so` the SDK dlopens; the SDK's default cache is per ABI revision (`abi/-`), and the revision is in both the path and the key prefix so an older SDK's cache is never restored. Measured for today's one pinned line (26.6, linux-amd64): **316 MiB on disk, ~0.06 GB stored**, and 85 MiB over the wire on a miss. Fetched by [`scripts/fetch-chtypes.sh`](../../scripts/fetch-chtypes.sh) (`--frozen`, refuses anything `chtypes.lock` doesn't name), which the CLI makes idempotent even on a cache hit — a manifest check, not a re-download. | +| chtypes artifact | `chtypes-abi6---` | unit / integration / e2e via `setup-env` (shared) | `~/.cache/chtypes/artifacts/abi6` — the pinned ClickHouse-version `.so` the SDK dlopens; the SDK's default cache is per ABI revision (`abi/-`), and the revision is in both the path and the key prefix so an older SDK's cache is never restored. Measured for the 26.6 line (linux-amd64), pinned before 26.8: **316 MiB on disk, ~0.06 GB stored**, and 85 MiB over the wire on a miss. Fetched by [`scripts/fetch-chtypes.sh`](../../scripts/fetch-chtypes.sh) (`--frozen`, refuses anything `chtypes.lock` doesn't name), which the CLI makes idempotent even on a cache hit — a manifest check, not a re-download. | | Go build objects (release) | `gobuild-v3---go-release-` | publish-dev (hand-rolled, not `setup-env`) | `~/.cache/go-build` from each `publish-dev` leg's own native `goreleaser build --single-target` (~0.5 GB per arch). `runner.arch` is load-bearing in the key: both legs (`ubuntu-latest`, `ubuntu-24.04-arm`) report `runner.os == 'Linux'`, so without it they'd overwrite the same entry every run — a guaranteed miss on one of them, forever. Same family and key inputs as the CI flavors; `-release` suffix because these objects share nothing with the native test-build flavors. Budget is now **2 arches × 2 generations** of a native single-target cache, not the old 1 × 2 generations of a single 3-target cross-compile cache — still narrower than the pre-chtypes 8-target matrix (Windows/FreeBSD/darwin-amd64 dropped — chtypes publishes no artifact for any of them). | | Go modules (release read) | `gomod-v1--` | nobody — **restore-only** | `publish-dev` reads `ci.yml`'s shared entry from `main`'s scope via `actions/cache/restore`, so its build isn't slowed by a cold module tree. No post-step save, so 0 GB of budget and no risk of a partial write to the shared key. | | CodeQL DB + deps | `codeql-dependencies-*`, `codeql-overlay-base-database-*` | GHAS default setup | **Not ours** — minted by GitHub's default CodeQL setup, not by any workflow in this repo, and not configurable here. ~0.4 GB. Listed so the budget arithmetic below is honest. | @@ -124,9 +124,9 @@ Never add a per-job copy of content that is a pure function of a lockfile — ke ## chtypes artifacts -`unit`, `integration` and `e2e` link the chtypes SDK (cgo dlopen of a per-ClickHouse-version `.so`/`.dylib`) and need the artifact for the line the test suite dials — today ClickHouse 26.6, matching `tests/integration/setup_test.go`'s pinned container. `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (job-level `env:` on all three) makes `typelayer.TestEngine` `t.Fatal()` if the artifact is missing instead of `t.Skip()`ing — CI must never quietly skip chtypes-backed tests. +`unit`, `integration` and `e2e` link the chtypes SDK (cgo dlopen of a per-ClickHouse-version `.so`/`.dylib`) and need the artifact for the line the test suite dials — today ClickHouse 26.8, matching `tests/integration/setup_test.go`'s pinned container. `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (job-level `env:` on all three) makes `typelayer.TestEngine` `t.Fatal()` if the artifact is missing instead of `t.Skip()`ing — CI must never quietly skip chtypes-backed tests. -`chtypes.lock` (repo root) pins the exact file + sha256 per platform/line; `scripts/fetch-chtypes.sh` wraps the SDK's own CLI with `--frozen --lock chtypes.lock`, so a fetch here can only install what the lock names, never the rolling `artifacts` release. `setup-env`'s `chtypes: "true"` input (see [Cache inventory](#cache-inventory)) restores `~/.cache/chtypes/artifacts/abi6` and always re-runs the fetch script afterward — cheap on a hit (a manifest check, not a re-download) and what turns a restored-but-unverified cache entry back into a hash-checked one every run. A lock is specific to the SDK's ABI revision: a build from another revision is never selected, so after an SDK bump that changes the revision (6 at v0.4.0) `--frozen` fails with `CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED` until the lock is regenerated the same way, and the `abi6` path and key prefix in `setup-env` move with it. Widening the pinned line set is a two-step: `scripts/fetch-chtypes.sh ` locally to update `chtypes.lock`, then add the line to `LOCK_LINES` in that script. Two hash mismatches are possible and they behave differently — don't read one as the other. **Upstream republished a pinned line under a new sha256**: `--frozen` refuses the artifact the lock does not name, and both `Dockerfile.goreleaser`'s fetch and CI's cache-miss fetch fail, until `chtypes.lock` is regenerated per platform (`go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch 26.6 --lock chtypes.lock --platform `, once each for `darwin-arm64`, `linux-amd64`, `linux-arm64`, without `--frozen`) — and `goreleaser-validate.yml`'s image job, which exercises this same fetch on every PR touching `chtypes.lock` or the release workflows, is what surfaces a republish at PR time rather than mid-release. **The cache holds a library the current lock no longer names** (a re-lock landed, so the exact key missed and `restore-keys` handed back the previous generation): this does *not* fail — measured, the CLI reports `is present but hashes … (want …) — replacing` and re-downloads, then the post-job save mints the new generation. So a re-lock costs one cold fetch per Go job on the first run and nothing after. +`chtypes.lock` (repo root) pins the exact file + sha256 per platform/line; `scripts/fetch-chtypes.sh` wraps the SDK's own CLI with `--frozen --lock chtypes.lock`, so a fetch here can only install what the lock names, never the rolling `artifacts` release. `setup-env`'s `chtypes: "true"` input (see [Cache inventory](#cache-inventory)) restores `~/.cache/chtypes/artifacts/abi6` and always re-runs the fetch script afterward — cheap on a hit (a manifest check, not a re-download) and what turns a restored-but-unverified cache entry back into a hash-checked one every run. A lock is specific to the SDK's ABI revision: a build from another revision is never selected, so after an SDK bump that changes the revision (6 at v0.4.0) `--frozen` fails with `CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED` until the lock is regenerated the same way, and the `abi6` path and key prefix in `setup-env` move with it. Widening the pinned line set is a two-step: `scripts/fetch-chtypes.sh ` locally to update `chtypes.lock`, then add the line to `LOCK_LINES` in that script. Two hash mismatches are possible and they behave differently — don't read one as the other. **Upstream republished a pinned line under a new sha256**: `--frozen` refuses the artifact the lock does not name, and both `Dockerfile.goreleaser`'s fetch and CI's cache-miss fetch fail, until `chtypes.lock` is regenerated per platform (`go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch 26.8 --lock chtypes.lock --platform `, once each for `darwin-arm64`, `linux-amd64`, `linux-arm64`, without `--frozen`) — and `goreleaser-validate.yml`'s image job, which exercises this same fetch on every PR touching `chtypes.lock` or the release workflows, is what surfaces a republish at PR time rather than mid-release. **The cache holds a library the current lock no longer names** (a re-lock landed, so the exact key missed and `restore-keys` handed back the previous generation): this does *not* fail — measured, the CLI reports `is present but hashes … (want …) — replacing` and re-downloads, then the post-job save mints the new generation. So a re-lock costs one cold fetch per Go job on the first run and nothing after. ## Timing (steady state, full pipeline) diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index 12a52fd3..050bb690 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -2602,7 +2602,7 @@ func TestIngest_TimestampsCanonicalized(t *testing.T) { req := ingestRequest(t, "events", map[string]any{ "name": "e", "ts": "2026-06-21 04:00:00", // zone-less ClickHouse-native form - "ts_ms": 1782014400500, // integer number = ClickHouse ticks at the column scale (ms here) + "ts_ms": 1782014400.5, // a JSON number is epoch seconds on 26.8; the fraction is sub-second }) w := httptest.NewRecorder() h.Handle(w, withTenant(req)) From 99a45296d26e033cf5fdabd340ed1b4ec6351c7a Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:29:05 -0400 Subject: [PATCH 19/70] refactor: delete the Go-side canonicalizer, row evaluator and compact encoder The type layer now judges timestamps, numeric row filters and insert checks, so nothing calls the Go reimplementations any more. Remove discovery's timestamp/validation code (and the per-column spec resolved at refresh), policy's in-memory row evaluator and numeric comparison, the LiteralValue marker, the unused WhereClause/WhereParams copies, the compact row encoder in ingest, and the stream hub's NumericSpecOf adapter. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/auth/auth_test.go | 8 +- internal/discovery/discovery.go | 21 +- internal/discovery/discovery_test.go | 5 +- internal/discovery/timestamp.go | 324 --------------- internal/discovery/timestamp_test.go | 309 --------------- internal/discovery/validation.go | 308 --------------- internal/discovery/validation_test.go | 358 ----------------- internal/ingest/compact.go | 52 --- internal/ingest/compact_test.go | 131 ------- internal/policy/canonical.go | 122 +----- internal/policy/numeric.go | 260 ------------ internal/policy/policy.go | 63 +-- internal/policy/policy_test.go | 130 +----- internal/policy/rowfilter.go | 294 -------------- internal/policy/rowfilter_test.go | 543 -------------------------- internal/stream/hub.go | 15 - internal/stream/hub_test.go | 2 +- 17 files changed, 66 insertions(+), 2879 deletions(-) delete mode 100644 internal/discovery/timestamp.go delete mode 100644 internal/discovery/timestamp_test.go delete mode 100644 internal/discovery/validation.go delete mode 100644 internal/discovery/validation_test.go delete mode 100644 internal/ingest/compact.go delete mode 100644 internal/ingest/compact_test.go delete mode 100644 internal/policy/numeric.go delete mode 100644 internal/policy/rowfilter.go delete mode 100644 internal/policy/rowfilter_test.go diff --git a/internal/auth/auth_test.go b/internal/auth/auth_test.go index 9e92cb9a..8e2d7fdc 100644 --- a/internal/auth/auth_test.go +++ b/internal/auth/auth_test.go @@ -182,8 +182,9 @@ func TestMiddleware_LargeIntegerClaim_ExactThroughPolicy(t *testing.T) { }} perms := policy.Evaluate(p, "viewer", "clicks", "select", c.claims) require.True(t, perms.Allowed) - assert.Equal(t, "`tenant_id` = ?", perms.Select.WhereClause) - assert.Equal(t, []any{"1234567890123456789"}, perms.Select.WhereParams) + where, whereParams := perms.Select.WhereSQL(nil) + assert.Equal(t, "`tenant_id` = ?", where) + assert.Equal(t, []any{"1234567890123456789"}, whereParams) } // TestMiddleware_NumericClaimSpelling_BindsCanonically: json.Number keeps the @@ -216,7 +217,8 @@ func TestMiddleware_NumericClaimSpelling_BindsCanonically(t *testing.T) { }} perms := policy.Evaluate(p, "viewer", "clicks", "select", c.claims) require.True(t, perms.Allowed) - assert.Equal(t, []any{tt.want}, perms.Select.WhereParams) + _, whereParams := perms.Select.WhereSQL(nil) + assert.Equal(t, []any{tt.want}, whereParams) }) } } diff --git a/internal/discovery/discovery.go b/internal/discovery/discovery.go index 08a09e23..529cb92c 100644 --- a/internal/discovery/discovery.go +++ b/internal/discovery/discovery.go @@ -59,12 +59,6 @@ type Column struct { // carries the ordinal itself so a caller holding a lone Column still knows // where it sits. Position uint64 `json:"position"` - - // tsSpec is a DateTime/DateTime64 column's canonicalization spec, resolved - // once at schema build (Refresh). nil for non-timestamp columns, hand-built - // Column literals, and unresolvable zones — CanonicalizeTimestamps passes - // those through untouched (fail-open, #372). - tsSpec *timestampSpec } // TableSchema holds the discovered schema for one ClickHouse table. Columns is @@ -299,8 +293,8 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { return fmt.Errorf("%w for tenant %s", ErrNoConnection, sr.tenant) } - // ClickHouse interprets zone-less timestamp strings in the server's default - // zone; canonicalization applies the same rule so the instant never changes (#372). + // The server's default zone, kept with the schemas so the type layer can + // check it against the zone its library was opened with. var tzName string if err := conn.QueryRow(ctx, "SELECT timezone()").Scan(&tzName); err != nil { return fmt.Errorf("query server timezone: %w", err) @@ -317,16 +311,6 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { return fmt.Errorf("query server version: %w", err) } - var serverTZ *time.Location - if loc, err := loadLocation(tzName); err == nil { - serverTZ = loc - } else { - // Unresolvable — warn, not fatal, and no UTC fallback (that could move - // instants). A nil server zone means zone-less values pass through. - slog.WarnContext(ctx, "cannot resolve server timezone; zone-less timestamps will pass through un-canonicalized", - "timezone", tzName, "error", err) - } - rows, err := conn.Query(ctx, `SELECT table, name, type, default_kind, default_expression, position FROM system.columns @@ -377,7 +361,6 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { var noDDL []string published := make([]*TableSchema, 0, len(tables)) for _, ts := range tables { - resolveTimestampSpecs(ctx, ts, serverTZ) ts.cacheInsertable() published = append(published, ts) if ts.DDL == "" { diff --git a/internal/discovery/discovery_test.go b/internal/discovery/discovery_test.go index 3456d341..862f349e 100644 --- a/internal/discovery/discovery_test.go +++ b/internal/discovery/discovery_test.go @@ -438,9 +438,8 @@ func newFakeRegistry(t *testing.T, errs []error) (*SchemaRegistry, *fakeConn) { } // TestRefresh_UnresolvableServerTimezone_NotFatal: an unresolvable server zone -// degrades to pass-through canonicalization (#372), never a failed refresh, -// and the registry still publishes the name verbatim: whoever consumes it -// decides what an unusable zone means. +// is never a failed refresh, and the registry still publishes the name +// verbatim: whoever consumes it decides what an unusable zone means. func TestRefresh_UnresolvableServerTimezone_NotFatal(t *testing.T) { t.Parallel() conn := &fakeConn{tz: "Not/AZone"} diff --git a/internal/discovery/timestamp.go b/internal/discovery/timestamp.go deleted file mode 100644 index dc0d8f81..00000000 --- a/internal/discovery/timestamp.go +++ /dev/null @@ -1,324 +0,0 @@ -package discovery - -import ( - "context" - "encoding/json" - "fmt" - "log/slog" - "math" - "strconv" - "strings" - "time" -) - -// Ingest canonicalizes every DateTime/DateTime64 column value to one wire form — -// RFC 3339 UTC (`2026-06-21T04:00:00Z`, fraction per column precision) — before -// the event is published (#372), so SSE subscribers see the same spelling -// /v1/query renders; fail-open, with fail-closed enforcement left to the stream -// row-filter (#381). The grammar is a strict subset of what ClickHouse's insert -// reads — mirrored per column kind and differential-tested against a live server -// (tests/integration/timestamp_canonicalization_test.go) — and everything else -// passes through verbatim, so ClickHouse is never second-guessed. - -// isTimestampType reports whether chType is a ClickHouse DateTime or DateTime64 -// (unwrapping Nullable/LowCardinality). Date/Date32 are excluded — day-precision, -// no zone or spelling ambiguity. -func isTimestampType(chType string) bool { - return strings.HasPrefix(unwrapType(chType), "DateTime") -} - -// CanonicalizeTimestamps rewrites every DateTime/DateTime64 column value in data -// to the canonical RFC 3339 UTC form, in place: zone-less values are read in the -// column's zone, else the server default — ClickHouse's own rule, so only the -// spelling changes, never the instant — and the fraction truncates to the -// column's precision to byte-match /v1/query. Everything else passes through -// verbatim (fail-open): absent/null values, unparseable values, columns without -// a spec, and instants outside the column kind's range (see rewritable). -func CanonicalizeTimestamps(schema *TableSchema, data map[string]any) { - for _, col := range schema.Columns { - spec := col.tsSpec - if spec == nil { - continue - } - v, ok := data[col.Name] - if !ok || v == nil { - continue - } - t, err := parseTimestamp(v, spec) - if err != nil || !spec.rewritable(t) { - continue - } - data[col.Name] = canonicalTimestamp(t, spec.precision) - } -} - -// TimeParser returns the mapping from one rendering of this DateTime/DateTime64 -// column's value to the instant ClickHouse would store: the same grammar and -// zone rule ingest canonicalization applies (parseTimestamp), the same range -// guard (rewritable — an out-of-range operand, which insert-time saturation -// would move, is refused), truncated to the column's precision exactly like the -// canonical wire form. nil when the column isn't a timestamp column with a -// resolved spec; such columns keep byte-equality semantics on the stream. The -// stream row-filter uses it (policy.ColumnSpec.ParseTime) so a filter constant -// in any accepted spelling — zone-less read in the column's zone, RFC 3339, -// Unix seconds — and the canonicalized payload compare as instants (#381), -// through one grammar that can't drift from ingest's. -func (c *Column) TimeParser() func(v any) (time.Time, bool) { - spec := c.tsSpec - if spec == nil { - return nil - } - unit := time.Second - for range spec.precision { - unit /= 10 - } - return func(v any) (time.Time, bool) { - t, err := parseTimestamp(v, spec) - if err != nil || !spec.rewritable(t) { - return time.Time{}, false - } - return t.Truncate(unit), true - } -} - -// timestampSpec is a timestamp column's precomputed canonicalization inputs: -// sub-second precision (0 for DateTime), the column kind (ClickHouse reads -// numbers and Unix-string fractions differently for DateTime64), and the zone -// for zone-less values (nil = unknown ⇒ only zone-explicit values canonicalize). -type timestampSpec struct { - precision int - isDT64 bool - loc *time.Location -} - -// resolveTimestampSpecs precomputes the timestampSpec of every DateTime/DateTime64 -// column in ts once per schema build, so the per-record ingest path parses no -// type strings and loads no zones. An unresolvable zone (no embedded tzdata — -// resolution needs the runtime's zone database) keeps a nil spec — warned, not -// fatal: those values pass through un-canonicalized. -func resolveTimestampSpecs(ctx context.Context, ts *TableSchema, serverTZ *time.Location) { - for i := range ts.Columns { - col := &ts.Columns[i] - if !isTimestampType(col.Type) { - continue - } - spec, err := resolveTimestampSpec(col.Type, serverTZ) - if err != nil { - slog.WarnContext(ctx, "cannot resolve timestamp column spec; its ingest values will pass through un-canonicalized", - "table", ts.Name, "column", col.Name, "type", col.Type, "error", err) - continue - } - col.tsSpec = &spec - } -} - -// resolveTimestampSpec extracts a DateTime/DateTime64 type's sub-second precision -// and time zone: `DateTime` / `DateTime('TZ')` / `DateTime64(P)` / -// `DateTime64(P, 'TZ')`. A type without an explicit zone takes serverTZ (possibly -// nil = unknown) — ClickHouse's own rule for zone-less strings. -func resolveTimestampSpec(chType string, serverTZ *time.Location) (timestampSpec, error) { - t := unwrapType(chType) - - var args []string - if open := strings.IndexByte(t, '('); open != -1 && strings.HasSuffix(t, ")") { - for arg := range strings.SplitSeq(t[open+1:len(t)-1], ",") { - args = append(args, strings.TrimSpace(arg)) - } - t = t[:open] - } - - isDT64 := t == "DateTime64" - var precision int - if isDT64 { - if len(args) == 0 { - return timestampSpec{}, fmt.Errorf("malformed type %q: DateTime64 requires a precision", chType) - } - p, err := strconv.Atoi(args[0]) - if err != nil || p < 0 || p > 9 { - return timestampSpec{}, fmt.Errorf("malformed type %q: bad precision %q", chType, args[0]) - } - precision = p - args = args[1:] - } - - if len(args) == 0 { - return timestampSpec{precision: precision, isDT64: isDT64, loc: serverTZ}, nil - } - name := strings.Trim(args[0], "'") - loc, err := loadLocation(name) - if err != nil { - return timestampSpec{}, fmt.Errorf("unknown time zone %q in type %q: %w", name, chType, err) - } - return timestampSpec{precision: precision, isDT64: isDT64, loc: loc}, nil -} - -func loadLocation(name string) (*time.Location, error) { - switch name { - case "Etc/UTC": - return time.UTC, nil - case "", "Local": - // time.LoadLocation reads "" as UTC and "Local" from the process - // environment — neither is a zone ClickHouse would declare. Treat both - // as unresolvable (nil spec, pass-through) instead of guessing. - return nil, fmt.Errorf("not an IANA zone name: %q", name) - } - return time.LoadLocation(name) -} - -// Rewrite bounds. ClickHouse saturates out-of-range instants spelling-dependently -// (the date clamps in the column zone keeping local time-of-day; DateTime64(9) -// even fails the insert past the Int64-nanosecond ceiling, a bound applied here -// to precision ≥7 conservatively), so no rewrite out there is safe — the -// producer's own spelling passes through and saturates as it did before #372. -// One-day margins keep zone arithmetic from straddling a bound. -var ( - dtMin = time.Unix(0, 0) - dtMax = time.Unix(4294967295-86400, 0) // UInt32 seconds ceiling, one day of margin - dt64Min = time.Date(1900, 1, 2, 0, 0, 0, 0, time.UTC) - dt64Max = time.Date(2299, 12, 30, 23, 59, 59, 0, time.UTC) - ns64Max = time.Date(2262, 4, 10, 23, 59, 59, 0, time.UTC) // Int64 ns ceiling (binds at precision 9; applied to ≥ 7) -) - -// rewritable reports whether t is safely inside the column kind's range — -// outside it the value passes through, so ClickHouse's saturation applies to -// the producer's spelling, never a rewritten one. -func (s *timestampSpec) rewritable(t time.Time) bool { - lo, hi := dtMin, dtMax - if s.isDT64 { - lo, hi = dt64Min, dt64Max - if s.precision >= 7 { - hi = ns64Max - } - } - return !t.Before(lo) && !t.After(hi) -} - -// parseTimestamp converts one ingested value into a time.Time: RFC 3339 -// ('.'-fraction only), zone-less `YYYY-MM-DD[ T]HH:MM:SS[.fraction]` / -// `YYYY-MM-DD` in the spec's zone (skipped when nil), and Unix seconds per -// parseUnixDigits and unixNumber — anything else errors and the caller passes -// it through. The grammar is deliberately a subset of ClickHouse best_effort's, -// instant-identical on that subset: when unsure what ClickHouse would read, -// this parser must fail, never guess. -func parseTimestamp(v any, spec *timestampSpec) (time.Time, error) { - switch x := v.(type) { - case string: - // ClickHouse has no ',' decimal separator (it would fight CSV), while - // Go's RFC3339Nano accepts one per ISO 8601. Reject up front so a - // ",999" fraction can't widen insertability. - if strings.ContainsRune(x, ',') { - return time.Time{}, fmt.Errorf( - "unrecognized timestamp %.64q (',' is not a fraction separator to ClickHouse)", x) - } - if t, err := time.Parse(time.RFC3339Nano, x); err == nil { - return t, nil - } - // Zone-less forms; Go's Parse accepts an input fraction after the seconds - // even when the layout carries none. - if spec.loc != nil { - for _, layout := range []string{"2006-01-02 15:04:05", "2006-01-02T15:04:05", "2006-01-02"} { - if t, err := time.ParseInLocation(layout, x, spec.loc); err == nil { - return t, nil - } - } - } - if t, ok := parseUnixDigits(x, spec.isDT64); ok { - return t, nil - } - return time.Time{}, fmt.Errorf( - "unrecognized timestamp %.64q (accepted: RFC 3339, 'YYYY-MM-DD[ T]HH:MM:SS[.fff]', 'YYYY-MM-DD', or 9-10 digit Unix seconds)", x) - case float64: - // Without json.Decoder.UseNumber a producer's integer is exact only - // through 2^53; past that the digits ClickHouse would read are already - // lost. Non-integers pass through — ClickHouse rejects them everywhere. - if x != math.Trunc(x) || math.Abs(x) > 1<<53 { - return time.Time{}, fmt.Errorf("timestamp number %v is not an exact integer", x) - } - return unixNumber(int64(x), spec) - case json.Number: - s := x.String() - if !isDigits(s) { - // A sign, '.', or exponent: not the integer epoch shape ClickHouse - // reads for DateTime columns — pass through for its verdict. - return time.Time{}, fmt.Errorf("timestamp number %q is not a non-negative integer", s) - } - n, err := strconv.ParseInt(s, 10, 64) - if err != nil { - return time.Time{}, fmt.Errorf("timestamp number %q: %w", s, err) - } - return unixNumber(n, spec) - default: - return time.Time{}, fmt.Errorf("timestamp must be a string or Unix-seconds number, got %T", v) - } -} - -// unixNumber mirrors how ClickHouse reads an integer JSON number: Unix seconds -// for DateTime, ticks at the column's scale for DateTime64 — reading a -// DateTime64 number as seconds would change the stored instant (1750478400 is -// a 1970 instant on a DateTime64(3); the ms epoch 1750478400500 is a valid -// 2025 one). Negative numbers pass through, and out-of-range results are -// caught by rewritable. -func unixNumber(n int64, spec *timestampSpec) (time.Time, error) { - if n < 0 { - return time.Time{}, fmt.Errorf("negative timestamp number %d", n) - } - if !spec.isDT64 { - return time.Unix(n, 0), nil - } - scale := int64(1) - for range spec.precision { - scale *= 10 - } - return time.Unix(n/scale, (n%scale)*(1_000_000_000/scale)), nil -} - -// parseUnixDigits reads Unix seconds in the one string shape ClickHouse -// best_effort does — 9–10 integer digits; other run lengths are its calendar -// forms (8 ⇒ YYYYMMDD, 14 ⇒ YYYYMMDDhhmmss) or its ms/µs/ns epochs (13/16/19) -// and pass through — with a '.' fraction honored only for DateTime64 targets -// (plain DateTime fails the row on it). The fraction is an exact decimal -// truncated at nine digits: ClickHouse truncates, never rounds, and a float64 -// round-trip would corrupt nanoseconds or roll across the second. -func parseUnixDigits(s string, isDT64 bool) (time.Time, bool) { - intPart, frac, hasFrac := strings.Cut(s, ".") - if len(intPart) < 9 || len(intPart) > 10 || !isDigits(intPart) { - return time.Time{}, false - } - if hasFrac && (!isDT64 || frac == "" || !isDigits(frac)) { - return time.Time{}, false - } - sec, _ := strconv.ParseInt(intPart, 10, 64) // 9-10 digits by construction - var ns int64 - if hasFrac { - f := frac - if len(f) > 9 { - f = f[:9] - } - ns, _ = strconv.ParseInt(f, 10, 64) // all digits by construction - // Zero-fill the fraction out to nanosecond scale. - for range 9 - len(f) { - ns *= 10 - } - } - return time.Unix(sec, ns), true -} - -func isDigits(s string) bool { - for i := 0; i < len(s); i++ { - if s[i] < '0' || s[i] > '9' { - return false - } - } - return true -} - -// canonicalTimestamp renders t in the canonical wire form: UTC RFC 3339, fraction -// truncated to the column's precision. RFC3339Nano trims trailing zeros exactly -// like /v1/query's transformRow, keeping the two read paths byte-identical. -func canonicalTimestamp(t time.Time, precision int) string { - unit := time.Second - for range precision { - unit /= 10 - } - return t.UTC().Truncate(unit).Format(time.RFC3339Nano) -} diff --git a/internal/discovery/timestamp_test.go b/internal/discovery/timestamp_test.go deleted file mode 100644 index 669ba620..00000000 --- a/internal/discovery/timestamp_test.go +++ /dev/null @@ -1,309 +0,0 @@ -package discovery - -import ( - "context" - "encoding/json" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/Wave-RF/WaveHouse/internal/tenant" -) - -func TestIsTimestampType(t *testing.T) { - t.Parallel() - - tests := []struct { - chType string - want bool - }{ - {"DateTime", true}, - {"DateTime('UTC')", true}, - {"DateTime64(3)", true}, - {"DateTime64(6, 'America/New_York')", true}, - {"Nullable(DateTime)", true}, - {"LowCardinality(Nullable(DateTime('UTC')))", true}, - {"Date", false}, // day-precision, no spelling ambiguity — deliberately excluded - {"Date32", false}, // as above - {"String", false}, - {"UInt64", false}, - } - - for _, tt := range tests { - t.Run(tt.chType, func(t *testing.T) { - t.Parallel() - assert.Equal(t, tt.want, isTimestampType(tt.chType)) - }) - } -} - -// tsSchema builds a one-column schema of the given ClickHouse type, so each case -// exercises exactly one column's canonicalization. -func tsSchema(colType string) *TableSchema { - return &TableSchema{Name: "t", Columns: []Column{{Name: "ts", Type: colType}}} -} - -func TestCanonicalizeTimestamps(t *testing.T) { - t.Parallel() - nyc, err := time.LoadLocation("America/New_York") - require.NoError(t, err) - - tests := []struct { - name string - colType string - serverTZ *time.Location - value any - want any - }{ - {"canonical passes through", "DateTime('UTC')", nil, "2026-06-21T04:00:00Z", "2026-06-21T04:00:00Z"}, - {"offset converts to Z", "DateTime('UTC')", nil, "2026-06-21T06:30:00+02:30", "2026-06-21T04:00:00Z"}, - {"naive space form, column zone", "DateTime('America/New_York')", nil, "2026-06-21 00:00:00", "2026-06-21T04:00:00Z"}, - {"naive T form, column zone", "DateTime('America/New_York')", nil, "2026-06-21T00:00:00", "2026-06-21T04:00:00Z"}, - {"naive form, Etc/UTC column zone", "DateTime('Etc/UTC')", nil, "2026-06-21 04:00:00", "2026-06-21T04:00:00Z"}, - {"naive form, server zone", "DateTime", nyc, "2026-06-21 00:00:00", "2026-06-21T04:00:00Z"}, - {"naive form, unknown server zone ⇒ passes through", "DateTime", nil, "2026-06-21 04:00:00", "2026-06-21 04:00:00"}, - {"offset form, unknown server zone still canonicalizes", "DateTime", nil, "2026-06-21T06:00:00+02:00", "2026-06-21T04:00:00Z"}, - {"date-only ⇒ midnight in zone", "DateTime('America/New_York')", nil, "2026-06-21", "2026-06-21T04:00:00Z"}, - {"unix seconds number", "DateTime('UTC')", nil, float64(1782014400), "2026-06-21T04:00:00Z"}, - {"unix seconds json.Number", "DateTime('UTC')", nil, json.Number("1782014400"), "2026-06-21T04:00:00Z"}, - {"unix seconds digit-string", "DateTime('UTC')", nil, "1782014400", "2026-06-21T04:00:00Z"}, - {"9-digit unix string", "DateTime('UTC')", nil, "999999999", "2001-09-09T01:46:39Z"}, - {"fractional unix digit-string", "DateTime64(3, 'UTC')", nil, "1782014400.5", "2026-06-21T04:00:00.5Z"}, - {"nanosecond fraction parsed exactly, not via float64", "DateTime64(9, 'UTC')", nil, "1782014400.123456789", "2026-06-21T04:00:00.123456789Z"}, - {"fraction digits beyond nine truncate, never round", "DateTime64(9, 'UTC')", nil, "1782014400.9999999995", "2026-06-21T04:00:00.999999999Z"}, - // Integer numbers are ClickHouse *ticks* at the column scale on DateTime64 - // — the ms epoch is the natural producer shape there, and an epoch-seconds - // number really is a 1970 instant (what the insert stores either way). - {"epoch-ms number is ticks on DateTime64(3)", "DateTime64(3, 'UTC')", nil, float64(1782014400000), "2026-06-21T04:00:00Z"}, - {"epoch-ms json.Number with sub-second ticks", "DateTime64(3, 'UTC')", nil, json.Number("1782014400123"), "2026-06-21T04:00:00.123Z"}, - {"epoch-seconds number on DateTime64(3) is a 1970 instant", "DateTime64(3, 'UTC')", nil, float64(1782014400), "1970-01-21T15:00:14.4Z"}, - {"number on DateTime64(0) is seconds (scale 1)", "DateTime64(0, 'UTC')", nil, float64(1782014400), "2026-06-21T04:00:00Z"}, - {"fraction truncated to column precision", "DateTime64(3, 'UTC')", nil, "2026-06-21T04:00:00.123456Z", "2026-06-21T04:00:00.123Z"}, - {"fraction truncated off a second-precision column", "DateTime('UTC')", nil, "2026-06-21 04:00:00.999", "2026-06-21T04:00:00Z"}, - {"trailing fractional zeros trimmed, like /v1/query", "DateTime64(3, 'UTC')", nil, "2026-06-21T04:00:00.120Z", "2026-06-21T04:00:00.12Z"}, - {"Nullable unwraps", "Nullable(DateTime('UTC'))", nil, "2026-06-21 04:00:00", "2026-06-21T04:00:00Z"}, - {"null left for ClickHouse", "Nullable(DateTime)", nil, nil, nil}, - {"String column untouched", "String", nil, "2026-06-21 04:00:00", "2026-06-21 04:00:00"}, - {"Date column untouched (excluded)", "Date", nil, "2026-06-21", "2026-06-21"}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - // The production path: specs resolved once at schema-build time. - schema := tsSchema(tt.colType) - resolveTimestampSpecs(t.Context(), schema, tt.serverTZ) - data := map[string]any{"ts": tt.value} - CanonicalizeTimestamps(schema, data) - assert.Equal(t, tt.want, data["ts"]) - }) - } -} - -// TestColumnTimeParser: the stream row-filter's per-column parser (#381) is the -// canonicalization grammar exactly — same spellings, zone rule, and Unix forms — -// truncated to the column's precision and bounded by the rewrite range, so a -// filter constant and a canonicalized payload always meet on the instant -// ClickHouse stores. -func TestColumnTimeParser(t *testing.T) { - t.Parallel() - utc4 := time.Date(2026, 6, 21, 4, 0, 0, 0, time.UTC) - - tests := []struct { - name string - colType string - value any - want time.Time - ok bool - }{ - {"canonical RFC 3339", "DateTime('UTC')", "2026-06-21T04:00:00Z", utc4, true}, - {"zone-less read in column zone", "DateTime('UTC')", "2026-06-21 04:00:00", utc4, true}, - {"explicit offset, same instant", "DateTime('UTC')", "2026-06-21T06:00:00+02:00", utc4, true}, - {"unix seconds string", "DateTime('UTC')", "1782014400", utc4, true}, - {"unix seconds number", "DateTime('UTC')", json.Number("1782014400"), utc4, true}, - {"fraction truncated to column precision", "DateTime64(1, 'UTC')", "2026-06-21T04:00:00.19Z", utc4.Add(100 * time.Millisecond), true}, - {"junk refused", "DateTime('UTC')", "not a timestamp", time.Time{}, false}, - {"out of range refused (insert-time saturation would move it)", "DateTime('UTC')", "2400-01-01T00:00:00Z", time.Time{}, false}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - schema := tsSchema(tt.colType) - resolveTimestampSpecs(t.Context(), schema, nil) - parse := schema.Columns[0].TimeParser() - require.NotNil(t, parse) - got, ok := parse(tt.value) - assert.Equal(t, tt.ok, ok) - if tt.ok { - assert.True(t, got.Equal(tt.want), "got %v, want %v", got, tt.want) - } - }) - } -} - -// TestColumnTimeParser_NilOrZoneLimited: only DateTime/DateTime64 columns with a -// resolved spec carry a parser — String/Date columns and hand-built literals -// return nil (byte-equality semantics on the stream). A timestamp column whose -// zone is unknown still parses zone-explicit forms but refuses zone-less strings: -// the zone would be a guess, and a guessed instant could move a row across a -// filter boundary. -func TestColumnTimeParser_NilOrZoneLimited(t *testing.T) { - t.Parallel() - schema := &TableSchema{Name: "t", Columns: []Column{ - {Name: "s", Type: "String"}, - {Name: "d", Type: "Date"}, - {Name: "ts", Type: "DateTime"}, - }} - resolveTimestampSpecs(t.Context(), schema, nil) - assert.Nil(t, schema.Columns[0].TimeParser(), "String column: no parser") - assert.Nil(t, schema.Columns[1].TimeParser(), "Date column: excluded from timestamp handling") - assert.Nil(t, tsSchema("DateTime").Columns[0].TimeParser(), "hand-built literal without spec resolution: no parser") - - unknownZone := schema.Columns[2].TimeParser() - require.NotNil(t, unknownZone, "zone-less DateTime with unknown server zone still has a (zone-explicit-only) parser") - _, ok := unknownZone("2026-06-21 04:00:00") - assert.False(t, ok, "zone-less string with unknown column zone: refused, never guessed") - got, ok := unknownZone("2026-06-21T04:00:00Z") - assert.True(t, ok) - assert.True(t, got.Equal(time.Date(2026, 6, 21, 4, 0, 0, 0, time.UTC))) -} - -// TestCanonicalizeTimestamps_NoPrecomputedSpec: a schema that skipped spec -// resolution (hand-built literals) passes through untouched. -func TestCanonicalizeTimestamps_NoPrecomputedSpec(t *testing.T) { - t.Parallel() - data := map[string]any{"ts": "2026-06-21 04:00:00"} - CanonicalizeTimestamps(tsSchema("DateTime"), data) - assert.Equal(t, "2026-06-21 04:00:00", data["ts"]) -} - -// TestCanonicalizeTimestamps_AbsentColumn: a column not in the payload (DEFAULT- -// filled by ClickHouse) is left absent, never invented. -func TestCanonicalizeTimestamps_AbsentColumn(t *testing.T) { - t.Parallel() - schema := &TableSchema{Name: "t", Columns: []Column{ - {Name: "ts", Type: "DateTime", HasDefault: true}, - {Name: "page", Type: "String"}, - }} - resolveTimestampSpecs(t.Context(), schema, time.UTC) - data := map[string]any{"page": "/home"} - CanonicalizeTimestamps(schema, data) - assert.Equal(t, map[string]any{"page": "/home"}, data) -} - -// TestCanonicalizeTimestamps_Unparseable_PassThrough: fail-open — unparseable -// values and unresolvable column specs pass through verbatim. -func TestCanonicalizeTimestamps_Unparseable_PassThrough(t *testing.T) { - t.Parallel() - - tests := []struct { - name string - colType string - value any - }{ - {"unrecognized string", "DateTime('UTC')", "banana"}, - {"non-finite numeric string is not an instant", "DateTime('UTC')", "NaN"}, - {"wrong value type", "DateTime('UTC')", true}, - {"unknown zone in type", "DateTime('Not/AZone')", "2026-06-21 04:00:00"}, - {"malformed DateTime64 precision", "DateTime64(x)", "2026-06-21 04:00:00"}, - // Digit-strings outside the 9–10 digit Unix shape mean calendar forms - // (or nothing) to ClickHouse best_effort — parsing them as Unix seconds - // would store a different instant than the insert (PR #402 review). - {"8-digit string is YYYYMMDD to ClickHouse", "DateTime('UTC')", "20260711"}, - {"12-digit string (ClickHouse rejects)", "DateTime('UTC')", "202607111500"}, - {"14-digit string is YYYYMMDDhhmmss to ClickHouse", "DateTime('UTC')", "20260711150000"}, - {"4-digit string is a year to ClickHouse", "DateTime('UTC')", "2026"}, - {"11-digit string", "DateTime('UTC')", "17504784000"}, - {"13-digit string is ClickHouse's ms epoch, not ours", "DateTime('UTC')", "1752278400000"}, - {"16-digit string is ClickHouse's µs epoch, not ours", "DateTime('UTC')", "1750478400123456"}, - {"scientific notation is not a timestamp", "DateTime('UTC')", "1e9"}, - {"negative digit-string", "DateTime('UTC')", "-100"}, - {"empty fraction", "DateTime('UTC')", "1750478400."}, - // ClickHouse consumes a fraction after a Unix epoch only for DateTime64 - // targets; on plain DateTime the leftover fraction fails the row. - {"fractional unix string on DateTime", "DateTime('UTC')", "1782014400.5"}, - // ClickHouse has no ',' decimal separator (Go's RFC3339Nano accepts one - // per ISO 8601) — rewriting would insert a row ClickHouse rejects raw. - {"comma fraction", "DateTime('UTC')", "2026-06-21T04:00:00,999Z"}, - {"comma fraction on DateTime64", "DateTime64(3, 'UTC')", "2026-06-21T04:00:00,9Z"}, - // ClickHouse rejects non-integer JSON numbers for every DateTime kind. - {"non-integer number", "DateTime64(3, 'UTC')", 1782014400.5}, - {"non-integer number on DateTime", "DateTime('UTC')", 1782014400.5}, - {"negative number", "DateTime('UTC')", float64(-100)}, - {"json.Number with exponent", "DateTime('UTC')", json.Number("1.5e9")}, - // Out of the column kind's range: ClickHouse saturates, and saturation is - // spelling-dependent (local time-of-day is kept while the date clamps), so - // no rewrite is safe — the raw spelling must be the one that saturates. - {"pre-epoch instant on DateTime", "DateTime('UTC')", "1960-01-01T00:00:00Z"}, - {"beyond UInt32 seconds on DateTime", "DateTime('UTC')", "2107-01-01T00:00:00Z"}, - {"number beyond UInt32 seconds on DateTime", "DateTime('UTC')", float64(4294967296)}, - {"beyond 2299 on DateTime64", "DateTime64(3, 'UTC')", "2300-06-30 12:30:00"}, - {"beyond the Int64-ns ceiling on DateTime64(9)", "DateTime64(9, 'UTC')", "2280-01-01T00:00:00Z"}, - // Valid RFC 3339 the subset deliberately omits (Go rejects :60). - {"leap-second spelling", "DateTime('UTC')", "2016-12-31T23:59:60Z"}, - // "" and "Local" are Go LoadLocation quirks (UTC / process env), not - // zone declarations — strict: unresolvable, pass through. - {"empty zone name in type", "DateTime('')", "2026-06-21 04:00:00"}, - {"Local zone in type", "DateTime('Local')", "2026-06-21 04:00:00"}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - schema := tsSchema(tt.colType) - resolveTimestampSpecs(t.Context(), schema, time.UTC) - data := map[string]any{"ts": tt.value} - CanonicalizeTimestamps(schema, data) - assert.Equal(t, tt.value, data["ts"]) - }) - } -} - -// TestResolveTimestampSpecs: timestamp columns get a spec (own zone, else server -// default); others don't; an unresolvable zone degrades to nil, not a failure. -func TestResolveTimestampSpecs(t *testing.T) { - t.Parallel() - nyc, err := time.LoadLocation("America/New_York") - require.NoError(t, err) - - schema := &TableSchema{Name: "t", Columns: []Column{ - {Name: "plain", Type: "DateTime"}, - {Name: "zoned", Type: "DateTime64(3, 'America/New_York')"}, - {Name: "page", Type: "String"}, - {Name: "broken", Type: "DateTime('Not/AZone')"}, - {Name: "etc_utc", Type: "DateTime('Etc/UTC')"}, - }} - resolveTimestampSpecs(t.Context(), schema, nyc) - - require.NotNil(t, schema.Columns[0].tsSpec) - assert.Equal(t, nyc, schema.Columns[0].tsSpec.loc, "zone-less column takes the server zone") - assert.False(t, schema.Columns[0].tsSpec.isDT64, "DateTime is not kind DateTime64") - require.NotNil(t, schema.Columns[1].tsSpec) - assert.Equal(t, nyc, schema.Columns[1].tsSpec.loc) - assert.Equal(t, 3, schema.Columns[1].tsSpec.precision) - assert.True(t, schema.Columns[1].tsSpec.isDT64, "numbers and unix fractions follow the DateTime64 rules") - assert.Nil(t, schema.Columns[2].tsSpec, "non-timestamp column gets no spec") - assert.Nil(t, schema.Columns[3].tsSpec, "unresolvable zone degrades to nil, not a failed build") - require.NotNil(t, schema.Columns[4].tsSpec) - assert.Same(t, time.UTC, schema.Columns[4].tsSpec.loc, "Etc/UTC maps to UTC without a tzdata lookup") - - // The degraded column passes through untouched — fail-open, never a rejection. - data := map[string]any{"broken": "2026-06-21 04:00:00"} - CanonicalizeTimestamps(schema, data) - assert.Equal(t, "2026-06-21 04:00:00", data["broken"]) -} - -// TestRefresh_PrecomputesSpecs: schema builds resolve timestamp specs with the -// server zone from SELECT timezone(), so registry consumers — production and -// the testutil mock-conn path alike — exercise the same precomputed path. -func TestRefresh_PrecomputesSpecs(t *testing.T) { - t.Parallel() - conn := &fakeConn{columns: []fakeColumn{{table: "t", name: "ts", chType: "DateTime", position: 1}}} - reg := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) - require.NoError(t, reg.Refresh(context.Background())) - col := reg.Get("t").Columns[0] - require.NotNil(t, col.tsSpec) - assert.Equal(t, time.UTC, col.tsSpec.loc) -} diff --git a/internal/discovery/validation.go b/internal/discovery/validation.go deleted file mode 100644 index 6fa1c556..00000000 --- a/internal/discovery/validation.go +++ /dev/null @@ -1,308 +0,0 @@ -package discovery - -import ( - "encoding/json" - "fmt" - "strconv" - "strings" -) - -// Validate checks that the given data matches the table schema. -// It rejects unknown fields and checks type compatibility. -func Validate(schema *TableSchema, data map[string]any) error { - colMap := make(map[string]Column, len(schema.Columns)) - for _, col := range schema.Columns { - colMap[col.Name] = col - } - - // TODO: the unknown field rejection can technically be controlled with `input_format_skip_unknown_fields = 1` I think, which would mean this is a false negative in some cases... - - // Reject unknown fields. - for key := range data { - col, ok := colMap[key] - if !ok { - return fmt.Errorf("unknown column %q for table %q", key, schema.Name) - } - // A computed column has no writable storage: ClickHouse refuses to - // insert a MATERIALIZED column and does not resolve an ALIAS one at - // all. The published row carries only insertable columns, so a value - // supplied for one of these would otherwise be dropped on the way out - // and the record would insert as though it had never been sent. - // Refuse it here instead, where the caller still hears about it. - if !col.IsInsertable() { - return fmt.Errorf("column %q of table %q is %s and cannot be inserted", - key, schema.Name, strings.ToLower(col.DefaultKind)) - } - } - - // TODO: I think clickhouse actually implicitly has defaults for strings, numbers etc like "" and 0 – so (if true) then omitting that column, even if the schema doesn't set that column to nullable or have a default, clickhouse will still set the implicit value if one – so technically we then shouldn't reject missing columns and this is a false negative? But that get's quite a bit messier... - - // Check type compatibility and required columns. - for _, col := range schema.Columns { - val, provided := data[col.Name] - if !provided { - if !col.IsNullable && !col.HasDefault { - return fmt.Errorf("missing required column %q for table %q", col.Name, schema.Name) - } - continue - } - if val == nil { - if !col.IsNullable && !col.HasDefault { - return fmt.Errorf("null value for non-nullable column %q", col.Name) - } - continue - } - if !isTypeCompatible(col.Type, val) { - return fmt.Errorf("type mismatch for column %q: cannot store %T in %s", col.Name, val, col.Type) - } - } - - return nil -} - -// unwrapType strips Nullable(...) and LowCardinality(...) modifiers (nested in any -// order, e.g. LowCardinality(Nullable(String))) down to the base ClickHouse type. -func unwrapType(chType string) string { - for { - if strings.HasPrefix(chType, "Nullable(") && strings.HasSuffix(chType, ")") { - chType = chType[9 : len(chType)-1] - continue - } - if strings.HasPrefix(chType, "LowCardinality(") && strings.HasSuffix(chType, ")") { - chType = chType[15 : len(chType)-1] - continue - } - return chType - } -} - -// isTypeCompatible checks whether a Go/JSON value can be stored in the given ClickHouse type. -func isTypeCompatible(chType string, val any) bool { - chType = unwrapType(chType) - - switch { - // String-compatible types accept Strings, Numbers (coerced), and Bools - case chType == "String", - strings.HasPrefix(chType, "FixedString("), - chType == "UUID": - switch val.(type) { - case string, float64, json.Number, bool: - return true - default: - return false - } - - // Dates/Times accept Strings and Numbers (Unix timestamps) - case strings.HasPrefix(chType, "DateTime"), - strings.HasPrefix(chType, "Date"): - switch val.(type) { - case string, float64, json.Number: - return true - default: - return false - } - - // Enums accept Strings (names) and Numbers (integer mappings) - case strings.HasPrefix(chType, "Enum8("), - strings.HasPrefix(chType, "Enum16("): - switch val.(type) { - case string, float64, json.Number: - return true - default: - return false - } - - // IPs accept Strings and Numbers (UInt32 representations) - case chType == "IPv4", chType == "IPv6": - switch val.(type) { - case string, float64, json.Number: - return true - default: - return false - } - - // Bools accept actual bools, numbers (0/1), and strings ("true"/"false") - case chType == "Bool": - switch val.(type) { - case bool, float64, json.Number, string: - return true - default: - return false - } - - // Numerics accept Numbers and Strings (to prevent JS precision loss) - case isNumericType(chType): - switch val.(type) { - case float64, json.Number, string: - return true - default: - return false - } - - // Array types accept JSON arrays. - case strings.HasPrefix(chType, "Array("): - _, ok := val.([]any) - return ok - - // Map types accept JSON objects. - case strings.HasPrefix(chType, "Map("): - _, ok := val.(map[string]any) - return ok - - // Tuple types accept JSON arrays or objects. - case strings.HasPrefix(chType, "Tuple("): - _, okArr := val.([]any) - _, okMap := val.(map[string]any) - return okArr || okMap - - default: - // Unknown type — accept any value and let ClickHouse validate. - return true - } -} - -// IsNumericType reports whether chType is a ClickHouse numeric type (integer, float, -// or decimal), unwrapping Nullable/LowCardinality modifiers first. The stream -// row-filter evaluator classifies such columns numeric-comparable, so ordering -// predicates (>, <) on numbers match ClickHouse (9 < 100). -func IsNumericType(chType string) bool { - return isNumericType(unwrapType(chType)) -} - -// IsStringType reports whether chType is a ClickHouse String (unwrapping -// Nullable/LowCardinality). For String columns, byte comparison IS ClickHouse -// comparison — equality and lexicographic order alike — so the stream row-filter -// evaluator can compare them exactly. FixedString is deliberately excluded: its -// stored values are zero-padded to the declared width, so a byte comparison of an -// ingested value against a filter constant would not match ClickHouse. -func IsStringType(chType string) bool { - return unwrapType(chType) == "String" -} - -// NumericStorage describes how a ClickHouse numeric column stores a value — -// the narrowing AND range the stream row-filter must apply to BOTH comparison -// operands so its verdicts match the query path, where ClickHouse narrows the -// stored value at insert and the bound constant at compare, and errors the -// query outright on a constant outside the column's range. Each family carries -// its parameters: Integer + IntBits/Unsigned for Int*/UInt* (exact within the -// width's range), FloatBits (32/64) for Float*, Precision+Scale for Decimal*. -type NumericStorage struct { - Integer bool - IntBits int - Unsigned bool - FloatBits int - Precision int - Scale int -} - -// NumericStorageOf classifies chType's numeric storage model, unwrapping -// Nullable/LowCardinality. ok=false for non-numeric types AND for a Decimal -// whose precision/scale cannot be parsed — the caller must then refuse numeric -// comparison rather than compare under a guessed model (fail closed). -// system.columns always reports the two-argument canonical Decimal(P, S) form -// (Decimal32(4) is stored as Decimal(9, 4)); the shorthand widths are handled -// anyway for robustness, and a bare single-argument Decimal(P) is refused -// rather than misread. -func NumericStorageOf(chType string) (NumericStorage, bool) { - chType = unwrapType(chType) - switch { - case !isNumericType(chType): - return NumericStorage{}, false - case chType == "Float32": - return NumericStorage{FloatBits: 32}, true - case chType == "Float64": - return NumericStorage{FloatBits: 64}, true - case strings.HasPrefix(chType, "Decimal"): - p, s, ok := decimalParams(chType) - if !ok { - return NumericStorage{}, false - } - return NumericStorage{Precision: p, Scale: s}, true - default: // isNumericType admits only Int*/UInt* beyond the cases above - bits, unsigned, ok := integerWidth(chType) - if !ok { - return NumericStorage{}, false - } - return NumericStorage{Integer: true, IntBits: bits, Unsigned: unsigned}, true - } -} - -// integerWidth reads the bit width and signedness from an Int*/UInt* type name. -func integerWidth(chType string) (bits int, unsigned bool, ok bool) { - rest, found := strings.CutPrefix(chType, "UInt") - if found { - unsigned = true - } else { - rest, found = strings.CutPrefix(chType, "Int") - if !found { - return 0, false, false - } - } - bits, err := strconv.Atoi(rest) - if err != nil { - return 0, false, false - } - switch bits { - case 8, 16, 32, 64, 128, 256: - return bits, unsigned, true - default: - return 0, false, false - } -} - -// decimalParams extracts (P, S) from Decimal(P, S) and the DecimalN(S) -// shorthands (Decimal32/64/128/256, whose precisions are fixed at 9/18/38/76). -// ClickHouse bounds them to 1 ≤ P ≤ 76 and 0 ≤ S ≤ P; anything outside that, -// malformed, or a single-argument Decimal(P) — whose lone number is a -// precision, not a scale — reports ok=false. -func decimalParams(chType string) (precision, scale int, ok bool) { - open := strings.IndexByte(chType, '(') - if open < 0 || !strings.HasSuffix(chType, ")") { - return 0, 0, false - } - args := strings.Split(chType[open+1:len(chType)-1], ",") - last, err := strconv.Atoi(strings.TrimSpace(args[len(args)-1])) - if err != nil { - return 0, 0, false - } - switch prefix := chType[:open]; prefix { - case "Decimal32": - precision = 9 - case "Decimal64": - precision = 18 - case "Decimal128": - precision = 38 - case "Decimal256": - precision = 76 - case "Decimal": - if len(args) != 2 { - return 0, 0, false - } - if precision, err = strconv.Atoi(strings.TrimSpace(args[0])); err != nil { - return 0, 0, false - } - default: - return 0, 0, false - } - scale = last - if precision < 1 || precision > 76 || scale < 0 || scale > precision { - return 0, 0, false - } - return precision, scale, true -} - -// isNumericType returns true for ClickHouse integer, float, and decimal types. -func isNumericType(chType string) bool { - switch { - case chType == "UInt8", chType == "UInt16", chType == "UInt32", chType == "UInt64", chType == "UInt128", chType == "UInt256": - return true - case chType == "Int8", chType == "Int16", chType == "Int32", chType == "Int64", chType == "Int128", chType == "Int256": - return true - case chType == "Float32", chType == "Float64": - return true - case strings.HasPrefix(chType, "Decimal"): - return true - default: - return false - } -} diff --git a/internal/discovery/validation_test.go b/internal/discovery/validation_test.go deleted file mode 100644 index 6028d232..00000000 --- a/internal/discovery/validation_test.go +++ /dev/null @@ -1,358 +0,0 @@ -package discovery - -import ( - "encoding/json" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func TestValidate(t *testing.T) { - t.Parallel() - - // A shared schema used across multiple tests - baseSchema := &TableSchema{ - Name: "clicks", - Columns: []Column{ - {Name: "user_id", Type: "String", IsNullable: false}, - {Name: "amount", Type: "Float64", IsNullable: false}, - {Name: "notes", Type: "Nullable(String)", IsNullable: true}, - {Name: "created_at", Type: "DateTime64(3, 'UTC')", HasDefault: true}, - }, - } - - tests := []struct { - name string - schema *TableSchema - data map[string]any - wantErr string // Substring to match in error; empty means success expected - }{ - { - name: "valid data all fields", - schema: baseSchema, - data: map[string]any{ - "user_id": "alice", - "amount": json.Number("42.5"), - "notes": "hello", - "created_at": "2025-01-01T00:00:00Z", - }, - }, - { - name: "unknown field rejected", - schema: baseSchema, - data: map[string]any{ - "user_id": "alice", - "amount": json.Number("42.5"), - "unknown": "value", - }, - wantErr: "unknown column", - }, - { - name: "type mismatch", - schema: baseSchema, - data: map[string]any{ - "user_id": json.Number("123"), // json.Number is valid for String - "amount": []any{"arrays", "fail"}, // invalid for Float64 - }, - wantErr: "type mismatch", - }, - { - name: "missing required column", - schema: baseSchema, - data: map[string]any{ - "user_id": "alice", - // amount is missing, not nullable, no default - }, - wantErr: "missing required column", - }, - { - name: "missing nullable column is allowed", - schema: baseSchema, - data: map[string]any{ - "user_id": "alice", - "amount": json.Number("42.5"), - // notes is omitted - }, - }, - { - name: "missing default column is allowed", - schema: baseSchema, - data: map[string]any{ - "user_id": "alice", - "amount": json.Number("42.5"), - // created_at is omitted - }, - }, - { - name: "nil for non-nullable rejected", - schema: baseSchema, - data: map[string]any{ - "user_id": nil, // non-nullable - "amount": json.Number("42.5"), - }, - wantErr: "null value for non-nullable", - }, - { - name: "nil for nullable allowed", - schema: baseSchema, - data: map[string]any{ - "user_id": "alice", - "amount": json.Number("42.5"), - "notes": nil, - }, - }, - { - name: "nil for non-nullable with default allowed", - schema: baseSchema, - data: map[string]any{ - "user_id": "alice", - "amount": json.Number("42.5"), - "created_at": nil, - }, - }, - { - name: "nil data triggers missing column", - schema: baseSchema, - data: nil, - wantErr: "missing required column", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - err := Validate(tt.schema, tt.data) - if tt.wantErr != "" { - require.Error(t, err) - assert.Contains(t, err.Error(), tt.wantErr) - } else { - assert.NoError(t, err) - } - }) - } -} - -func TestIsTypeCompatible(t *testing.T) { - t.Parallel() - - tests := []struct { - name string - chType string - val any - want bool - }{ - // String and Date types - {"String accepts string", "String", "hello", true}, - {"String accepts float64", "String", 42.0, true}, - {"String accepts json.Number", "String", json.Number("42"), true}, - {"String accepts bool", "String", true, true}, - {"String rejects array", "String", []any{}, false}, - {"DateTime accepts string", "DateTime64(3, 'UTC')", "2025-01-01", true}, - {"DateTime accepts number", "DateTime64", json.Number("1716570889"), true}, - - // Numeric types - {"UInt64 accepts float64", "UInt64", 42.0, true}, - {"UInt64 accepts json.Number", "UInt64", json.Number("1234567890"), true}, - {"UInt64 accepts string", "UInt64", "1234567890", true}, - {"Decimal accepts string", "Decimal(18,4)", "42.5000", true}, - {"Float64 rejects bool", "Float64", true, false}, - - // Bool - {"Bool accepts bool", "Bool", true, true}, - {"Bool accepts float64", "Bool", 1.0, true}, - {"Bool accepts json.Number", "Bool", json.Number("0"), true}, - {"Bool accepts string", "Bool", "true", true}, - {"Bool rejects array", "Bool", []any{}, false}, - - // Enums & IPs - {"Enum accepts string", "Enum16('a'=1,'b'=2)", "a", true}, - {"Enum accepts number", "Enum8('a'=1)", json.Number("1"), true}, - {"IPv4 accepts string", "IPv4", "192.168.1.1", true}, - {"IPv4 accepts number", "IPv4", json.Number("3232235777"), true}, - - // Complex Types - {"Array accepts slice", "Array(String)", []any{"a", "b"}, true}, - {"Array rejects string", "Array(String)", "not-an-array", false}, - {"Map accepts map", "Map(String, String)", map[string]any{"k": "v"}, true}, - {"Tuple accepts slice", "Tuple(String, Int32)", []any{"a", 1.0}, true}, - {"Tuple accepts map", "Tuple(a String, b Int32)", map[string]any{"a": "x", "b": 1.0}, true}, - - // Modifiers (Nullable / LowCardinality) - {"Nullable accepts valid", "Nullable(String)", "hello", true}, - {"LowCardinality accepts valid", "LowCardinality(String)", "hello", true}, - {"Nested modifiers unwrapped 1", "LowCardinality(Nullable(String))", "hello", true}, - {"Nested modifiers unwrapped 2", "Nullable(LowCardinality(String))", "hello", true}, - {"Nested modifiers reject invalid", "LowCardinality(Nullable(UInt64))", []any{}, false}, - - // Fallback - {"Unknown type accepts everything", "SomeFutureType", []any{"sure"}, true}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - got := isTypeCompatible(tt.chType, tt.val) - assert.Equal(t, tt.want, got, "chType=%q, val=%T(%v)", tt.chType, tt.val, tt.val) - }) - } -} - -func TestIsNumericType(t *testing.T) { - t.Parallel() - - tests := []struct { - chType string - want bool - }{ - {"UInt8", true}, - {"UInt256", true}, - {"Int16", true}, - {"Int128", true}, - {"Float32", true}, - {"Float64", true}, - {"Decimal(10,2)", true}, - {"Decimal128(5)", true}, - {"String", false}, - {"Bool", false}, - } - - for _, tt := range tests { - t.Run(tt.chType, func(t *testing.T) { - t.Parallel() - assert.Equal(t, tt.want, isNumericType(tt.chType)) - }) - } -} - -// TestIsNumericType_Exported checks the exported wrapper the stream row-filter uses: -// it must unwrap Nullable/LowCardinality (in any nesting) before classifying, so a -// Nullable(UInt64) column still compares numerically. -func TestIsNumericType_Exported(t *testing.T) { - t.Parallel() - - tests := []struct { - chType string - want bool - }{ - {"UInt64", true}, - {"Nullable(UInt64)", true}, - {"LowCardinality(Int32)", true}, - {"LowCardinality(Nullable(Float64))", true}, - {"Decimal(10,2)", true}, - {"String", false}, - {"Nullable(String)", false}, - {"LowCardinality(String)", false}, - {"DateTime", false}, - } - - for _, tt := range tests { - t.Run(tt.chType, func(t *testing.T) { - t.Parallel() - assert.Equal(t, tt.want, IsNumericType(tt.chType)) - }) - } -} - -// TestIsStringType: only String (under any Nullable/LowCardinality wrapping) -// qualifies — byte comparison is ClickHouse comparison for it. FixedString is -// excluded on purpose (zero-padded storage), as is everything whose text form is -// not canonical (UUID, Enum, DateTime, Bool). -func TestIsStringType(t *testing.T) { - t.Parallel() - - tests := []struct { - chType string - want bool - }{ - {"String", true}, - {"Nullable(String)", true}, - {"LowCardinality(String)", true}, - {"LowCardinality(Nullable(String))", true}, - {"FixedString(16)", false}, - {"UUID", false}, - {"Enum8('a' = 1)", false}, - {"DateTime", false}, - {"Bool", false}, - {"UInt64", false}, - } - - for _, tt := range tests { - t.Run(tt.chType, func(t *testing.T) { - t.Parallel() - assert.Equal(t, tt.want, IsStringType(tt.chType)) - }) - } -} - -// TestNumericStorageOf pins the storage classification the stream row-filter -// narrows comparisons with: integer family exact at any width, float bit -// widths, Decimal scale extraction across every declaration form, wrappers -// unwrapped, and ok=false for non-numerics and for a Decimal whose scale can't -// be parsed — the caller must refuse comparison rather than guess a model. -func TestNumericStorageOf(t *testing.T) { - t.Parallel() - tests := []struct { - chType string - want NumericStorage - ok bool - }{ - {"UInt64", NumericStorage{Integer: true, IntBits: 64, Unsigned: true}, true}, - {"Int256", NumericStorage{Integer: true, IntBits: 256}, true}, - {"Nullable(UInt32)", NumericStorage{Integer: true, IntBits: 32, Unsigned: true}, true}, - {"Float32", NumericStorage{FloatBits: 32}, true}, - {"LowCardinality(Nullable(Float64))", NumericStorage{FloatBits: 64}, true}, - {"Decimal(10, 2)", NumericStorage{Precision: 10, Scale: 2}, true}, - {"Decimal(10,2)", NumericStorage{Precision: 10, Scale: 2}, true}, - {"Decimal(2, 2)", NumericStorage{Precision: 2, Scale: 2}, true}, - {"Decimal32(4)", NumericStorage{Precision: 9, Scale: 4}, true}, - {"Decimal64(0)", NumericStorage{Precision: 18, Scale: 0}, true}, - {"Decimal256(76)", NumericStorage{Precision: 76, Scale: 76}, true}, - {"Decimal", NumericStorage{}, false}, - {"Decimal(10)", NumericStorage{}, false}, - {"Decimal(10, -1)", NumericStorage{}, false}, - {"Decimal(10, 77)", NumericStorage{}, false}, - {"String", NumericStorage{}, false}, - {"DateTime", NumericStorage{}, false}, - {"Bool", NumericStorage{}, false}, - } - for _, tt := range tests { - t.Run(tt.chType, func(t *testing.T) { - t.Parallel() - got, ok := NumericStorageOf(tt.chType) - assert.Equal(t, tt.ok, ok) - assert.Equal(t, tt.want, got) - }) - } -} - -// TestValidate_RejectsSuppliedComputedColumn: a record naming a MATERIALIZED or -// ALIAS column must be refused where the caller still hears about it. The -// published row carries only insertable columns, so without this the value -// would be silently dropped and the record would insert as though it had never -// been sent — the failure mode that made this class invisible. -func TestValidate_RejectsSuppliedComputedColumn(t *testing.T) { - t.Parallel() - schema := &TableSchema{Name: "clicks", Columns: []Column{ - {Name: "id", Type: "UInt64"}, - {Name: "mat", Type: "String", DefaultKind: "MATERIALIZED", HasDefault: true}, - {Name: "ali", Type: "UInt64", DefaultKind: "ALIAS", HasDefault: true}, - }} - - for _, tt := range []struct{ name, col, want string }{ - {"materialized", "mat", "materialized and cannot be inserted"}, - {"alias", "ali", "alias and cannot be inserted"}, - } { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - err := Validate(schema, map[string]any{"id": float64(1), tt.col: "x"}) - require.Error(t, err) - assert.Contains(t, err.Error(), tt.col) - assert.Contains(t, err.Error(), tt.want) - }) - } - - // Omitting them is the normal case and must still pass: they are not - // "missing required columns", they are computed by the server. - require.NoError(t, Validate(schema, map[string]any{"id": float64(1)})) -} diff --git a/internal/ingest/compact.go b/internal/ingest/compact.go deleted file mode 100644 index e3ff6e75..00000000 --- a/internal/ingest/compact.go +++ /dev/null @@ -1,52 +0,0 @@ -package ingest - -import ( - "bytes" - "encoding/json" - "fmt" - - "github.com/Wave-RF/WaveHouse/internal/discovery" -) - -// EncodeCompactRow renders one record as a single JSONCompactEachRow line: a -// JSON array carrying exactly one value per INSERTABLE column, in declaration order -// (discovery orders TableSchema.Columns by system.columns.position). A column -// the record does not carry encodes as null. For a NON-nullable column with a -// default the insert turns that back into the default -// (input_format_null_as_default); on a NULLABLE column ClickHouse stores the -// NULL, because only an ABSENT key ever took the default and a positional row -// cannot express absence. worker.go pins -// input_format_null_as_default=1 explicitly, alongside date_time_input_format, -// so a server-default change cannot silently alter either. -// The result has no trailing newline — the caller joins lines. -// -// This is serialization ONLY. It performs no validation and makes no decision -// about a value: schema validation upstream has already rejected unknown keys -// and unacceptable types, so there is nothing here to reject and nothing to -// coerce. Record values arrive json.Number-preserving (every decoder on the -// ingest path sets UseNumber), and json.Marshal writes a json.Number as its -// exact digits, so a 64-bit id past 2^53 keeps every one of them. -// -// Transitional: replaced by chtypes RowsExport; serialization only, never add -// rules here. -func EncodeCompactRow(orderedCols []discovery.Column, record map[string]any) (json.RawMessage, error) { - var buf bytes.Buffer - buf.WriteByte('[') - for i, col := range orderedCols { - if i > 0 { - buf.WriteByte(',') - } - v, ok := record[col.Name] - if !ok { - buf.WriteString("null") - continue - } - b, err := json.Marshal(v) - if err != nil { - return nil, fmt.Errorf("encode column %q: %w", col.Name, err) - } - buf.Write(b) - } - buf.WriteByte(']') - return json.RawMessage(buf.Bytes()), nil -} diff --git a/internal/ingest/compact_test.go b/internal/ingest/compact_test.go deleted file mode 100644 index 40ac676e..00000000 --- a/internal/ingest/compact_test.go +++ /dev/null @@ -1,131 +0,0 @@ -package ingest - -import ( - "encoding/json" - "math" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/Wave-RF/WaveHouse/internal/discovery" -) - -func cols(names ...string) []discovery.Column { - out := make([]discovery.Column, 0, len(names)) - for i, n := range names { - out = append(out, discovery.Column{Name: n, Type: "String", Position: uint64(i + 1)}) - } - return out -} - -// TestEncodeCompactRow_DeclarationOrder: the array follows the SCHEMA's order, -// not the record's — a Go map has none, which is the whole reason the column -// list travels separately from the row. -func TestEncodeCompactRow_DeclarationOrder(t *testing.T) { - t.Parallel() - line, err := EncodeCompactRow(cols("page", "button", "country"), map[string]any{ - "country": "US", - "page": "/home", - "button": "signup", - }) - require.NoError(t, err) - assert.JSONEq(t, `["/home","signup","US"]`, string(line)) - assert.Equal(t, `["/home","signup","US"]`, string(line), "positional output must be byte-exact, not merely equivalent") -} - -// TestEncodeCompactRow_MissingColumnIsNull: a column the record omits holds a -// position in the array — dropping it would shift every value after it. -func TestEncodeCompactRow_MissingColumnIsNull(t *testing.T) { - t.Parallel() - line, err := EncodeCompactRow(cols("page", "button", "country"), map[string]any{ - "page": "/home", - "country": "US", - }) - require.NoError(t, err) - assert.Equal(t, `["/home",null,"US"]`, string(line)) -} - -// TestEncodeCompactRow_ExplicitNullAndMissingAgree: a column present as JSON -// null encodes the same as one that is absent, so the insert path treats them -// alike. -func TestEncodeCompactRow_ExplicitNullAndMissingAgree(t *testing.T) { - t.Parallel() - explicit, err := EncodeCompactRow(cols("page", "button"), map[string]any{"page": "/a", "button": nil}) - require.NoError(t, err) - absent, err := EncodeCompactRow(cols("page", "button"), map[string]any{"page": "/a"}) - require.NoError(t, err) - assert.Equal(t, string(absent), string(explicit)) -} - -// TestEncodeCompactRow_NumberFidelity: a json.Number keeps its exact digits, so -// a 64-bit id past 2^53 survives the round trip a float64 would round off. -func TestEncodeCompactRow_NumberFidelity(t *testing.T) { - t.Parallel() - const bigID = "9007199254740993" // 2^53 + 1: not representable as a float64 - line, err := EncodeCompactRow(cols("id", "ratio"), map[string]any{ - "id": json.Number(bigID), - "ratio": json.Number("1.500"), - }) - require.NoError(t, err) - assert.Equal(t, `[9007199254740993,1.500]`, string(line), - "json.Number writes its exact digits, trailing zeros and all") - - // The same value decoded as a float64 would not survive — the guard the - // UseNumber decoders on the ingest path exist for. - lossy, err := EncodeCompactRow(cols("id"), map[string]any{"id": float64(9007199254740993)}) - require.NoError(t, err) - assert.NotEqual(t, "["+bigID+"]", string(lossy)) -} - -// TestEncodeCompactRow_EmptySchema: no columns is an empty array, never a bare -// or malformed line. -func TestEncodeCompactRow_EmptySchema(t *testing.T) { - t.Parallel() - for _, record := range []map[string]any{nil, {}, {"stray": 1}} { - line, err := EncodeCompactRow(nil, record) - require.NoError(t, err) - assert.Equal(t, `[]`, string(line)) - } -} - -// TestEncodeCompactRow_NoTrailingNewline: the caller joins lines, so a line -// must not carry its own terminator. -func TestEncodeCompactRow_NoTrailingNewline(t *testing.T) { - t.Parallel() - line, err := EncodeCompactRow(cols("page"), map[string]any{"page": "/a"}) - require.NoError(t, err) - assert.NotContains(t, string(line), "\n") - assert.Equal(t, byte(']'), line[len(line)-1]) -} - -// TestEncodeCompactRow_StructuredAndUnicodeValues: arrays, maps, and non-ASCII -// text pass through as the JSON they are — the encoder judges no value. -func TestEncodeCompactRow_StructuredAndUnicodeValues(t *testing.T) { - t.Parallel() - line, err := EncodeCompactRow(cols("tags", "attrs", "label"), map[string]any{ - "tags": []any{"a", "b"}, - "attrs": map[string]any{"k": "v"}, - "label": "héllo · 世界", - }) - require.NoError(t, err) - - var decoded []any - require.NoError(t, json.Unmarshal(line, &decoded)) - require.Len(t, decoded, 3) - assert.Equal(t, []any{"a", "b"}, decoded[0]) - assert.Equal(t, map[string]any{"k": "v"}, decoded[1]) - assert.Equal(t, "héllo · 世界", decoded[2]) -} - -// TestEncodeCompactRow_UnmarshalableValue: a value encoding/json cannot render -// is an error naming the column, not a silently dropped or shifted field. -func TestEncodeCompactRow_UnmarshalableValue(t *testing.T) { - t.Parallel() - _, err := EncodeCompactRow(cols("page", "score"), map[string]any{ - "page": "/a", - "score": math.Inf(1), // JSON has no infinity - }) - require.Error(t, err) - assert.Contains(t, err.Error(), `"score"`) -} diff --git a/internal/policy/canonical.go b/internal/policy/canonical.go index 1f91776f..b3714ef9 100644 --- a/internal/policy/canonical.go +++ b/internal/policy/canonical.go @@ -9,28 +9,12 @@ import ( "strings" ) -// This file is the policy engine's ONE rendering layer for comparison operands: -// every value a row-filter or insert-check compares — a JWT claim, a -// policy-authored literal, an ingested payload value — is rendered here before -// any comparison or SQL bind, so the two read surfaces can't disagree on what a -// value "is". Three entry points, one per operand source: -// -// CanonicalScalar — decoded claim values (Evaluate's template resolution) -// and the insert-check comparison's two sides (internal/api) -// CanonicalNumericLiteral — policy-authored literals ("1.0" → "1"), behind the -// json.Valid grammar gate -// numericCanonical — ingested payload values on the stream's row-filter -// path (string / json.Number / float64), routed through -// the two above -// -// All three converge on one canonical decimal form (canonicalDecimal, bounded -// by maxCanonicalDigits): exact at any width, positional (never an exponent — +// CanonicalScalar is the policy engine's one rendering for comparison operands +// that come from a JWT claim: every value a row-filter binds is rendered here +// before any SQL bind, so the read surfaces can't disagree on what a value +// "is". It renders exact canonical decimal form (canonicalDecimal, bounded by +// maxCanonicalDigits): exact at any width, positional (never an exponent — // "1e-3" renders as "0.001"), no leading/trailing zeros, "-0" folded to "0". -// Those invariants are what numeric.go's digit-string comparison -// (compareCanonicalDecimals) and storage-domain narrowing (NumericSpec.compare) -// rely on. scalarString is the deliberate exception: the raw byte rendering for -// ColumnText/ColumnOpaque comparison, where the payload's own spelling IS the -// compared value. // maxCanonicalDigits bounds both a numeric literal's digit count and its // exact decimal expansion. Big-integer parsing is superlinear in digit count @@ -40,16 +24,6 @@ import ( // any real id — uint256 is 78. const maxCanonicalDigits = 100 -// maxNumericOperandChars is the O(1) length pre-gate on numeric comparison -// operands, checked before ANY scan of the value. It is verdict-preserving: a -// JSON number literal carries at most four non-digit bytes (a sign, a decimal -// point, an exponent marker and its sign), so anything longer already fails -// the canonical digit bound (maxCanonicalDigits) — but proving that inside the -// canonical gate costs a full json.Valid pass plus a digit count over a -// client-controlled value, per subscriber per event on the fan-out goroutine. -// The gate refuses first, without reading the bytes. -const maxNumericOperandChars = maxCanonicalDigits + 4 - // CanonicalScalar renders a decoded JSON value as the canonical string the // policy layer binds and compares, reporting ok=false for values with no such // form: null, objects, and arrays. A structured value is never a sensible @@ -112,30 +86,6 @@ func CanonicalScalar(v any) (string, bool) { } } -// LiteralValue marks an insert-check required value the policy author wrote -// as a placeholder-free literal (Evaluate). A literal carries no JSON type — -// "1.0" means the number 1 to a numeric column and the three-character text -// to a String column — so the check comparison (internal/api) accepts its -// numeric reading as well as its spelling. The type is the gate: a -// claim-derived value is never wrapped, so a string-typed claim keeps strict -// canonical equality and can't gain a numeric reading it didn't have. Only -// CheckClauses carries this type; read filters bind plain strings. -type LiteralValue string - -// CanonicalNumericLiteral renders a policy-authored literal that spells a -// JSON number in canonical decimal form ("1.0" → "1"), reporting ok=false for -// everything else. The json.Valid gate keeps this to spellings JSON itself -// can produce: big.Int would also take "+5" or "007", readings no decoded -// claim or payload value ever has. It canonicalizes nothing at resolve time — -// the literal still binds and auto-injects exactly as written; only the check -// comparison consults this second reading. -func CanonicalNumericLiteral(s string) (string, bool) { - if !json.Valid([]byte(s)) { - return "", false - } - return CanonicalScalar(json.Number(s)) -} - // canonicalDecimal renders a non-integer JSON number literal (one carrying a // fraction or exponent) as its exact canonical decimal string: "1.0" → "1", // "2.50" → "2.5", "1e3" → "1000", "25e-4" → "0.0025", every digit preserved @@ -209,65 +159,3 @@ func canonicalDecimal(lit string) (string, bool) { } return sign + out, true } - -// numericCanonical renders a payload value as the canonical decimal form the -// numeric comparison consumes, ok=false for anything that is not a number a -// ClickHouse numeric column could have stored: booleans, structured values and -// null, spellings outside the JSON number grammar ("Inf", "NaN", "0x1f", -// "007"), values whose exact digits were lost upstream (float64 at/past 2^53), -// and anything past the canonical digit bound. String and json.Number inputs -// take the claim side's own gates (CanonicalNumericLiteral / CanonicalScalar), -// so the payload and constant sides can never disagree on what counts as a -// number or how it is spelled. -func numericCanonical(v any) (string, bool) { - switch x := v.(type) { - case string: - if len(x) > maxNumericOperandChars { - return "", false - } - return CanonicalNumericLiteral(x) - case json.Number: - if len(x) > maxNumericOperandChars { - return "", false - } - return CanonicalScalar(x) - case float64: - // CanonicalScalar applies the 2^53 exactness guard and renders - // positionally; the literal gate then re-canonicalizes the one - // rendering FormatFloat emits that canonical form forbids ("-0"). - s, ok := CanonicalScalar(x) - if !ok { - return "", false - } - return CanonicalNumericLiteral(s) - default: - return "", false - } -} - -// scalarString renders a JSON-decoded scalar as the exact BYTES compared under -// ColumnText and ColumnOpaque — deliberately the payload's raw spelling, never -// a canonical form: a String column stores the payload text verbatim, so byte -// comparison against it must use that spelling (canonicalizing "1.0" to "1" -// here would move equality away from what ClickHouse stores). Numeric coercion -// deliberately does NOT live here — the ColumnNumeric arm routes both operands -// through the claim side's canonical machinery (numericCanonical). Non-scalars -// (arrays, objects, null) return ok=false so the predicate fails closed rather -// than guessing. The float64 case serves callers that decoded without -// UseNumber (the stream itself always does); -1 precision emits the shortest -// round-trip form without an exponent, so integer IDs read back as "123", not -// "1.23e+02". -func scalarString(v any) (string, bool) { - switch x := v.(type) { - case string: - return x, true - case json.Number: - return string(x), true - case float64: - return strconv.FormatFloat(x, 'f', -1, 64), true - case bool: - return strconv.FormatBool(x), true - default: - return "", false - } -} diff --git a/internal/policy/numeric.go b/internal/policy/numeric.go deleted file mode 100644 index 1a27017b..00000000 --- a/internal/policy/numeric.go +++ /dev/null @@ -1,260 +0,0 @@ -package policy - -import ( - "math/big" - "strconv" - "strings" -) - -// This file compares canonical decimal forms (canonical.go's output) the way -// the column that stores them would: compareCanonicalDecimals is the exact -// digit-string ordering, and NumericSpec narrows both operands into the -// column's STORAGE domain first — the same narrowing ClickHouse applies to the -// stored value at insert and to the filter constant at compare — so the -// stream's in-memory verdict can't drift from the query path's SQL verdict. - -// NumericFamily classifies how a ClickHouse numeric column stores a value — -// the narrowing the row-filter comparison must apply to BOTH operands so its -// verdict matches the query path, where ClickHouse narrows the stored value at -// insert AND the filter constant at compare. The zero value is NumericNone: no -// storage model, every comparison refused — the same fail-closed zero-value -// contract as ColumnOpaque, so a future numeric type nobody classified can -// never be compared under the wrong model. -type NumericFamily uint8 - -const ( - NumericNone NumericFamily = iota // unclassified: refuse, fail closed - NumericInteger // Int*/UInt*: exact at any width - NumericFloat // Float32/Float64: IEEE rounding at Bits - NumericDecimal // Decimal*: truncation at Scale -) - -// NumericSpec is a numeric column's storage model. Bits is the bit width -// (float width for NumericFloat, integer width for NumericInteger); Unsigned -// marks UInt* (NumericInteger only); Precision and Scale are the stored total -// and fractional digit counts (NumericDecimal only, 1 ≤ Precision ≤ 76). -type NumericSpec struct { - Family NumericFamily - Bits int - Unsigned bool - Precision int - Scale int -} - -// intBounds holds each ClickHouse integer width's inclusive decimal bounds as -// canonical-form strings, computed once. A constant outside the width is not -// reliably modelable — on one and the same release, ClickHouse was measured to -// ERROR the comparison (a negative literal against an unsigned column: the -// role reads no rows), to PROMOTE and compare mathematically ('256' against a -// UInt8), and to WRAP at a width boundary ('9223372036854775808' against an -// Int64 compares as −2^63, where exact-precision comparison would ADMIT the -// −2^63 rows SQL hides under !=). Refusing out-of-range operands is the one -// rule safe under all three behaviors; the cost is availability on bounds no -// in-range data could ever satisfy differently. -var intBounds = func() map[int]struct{ sMin, sMax, uMax string } { - m := make(map[int]struct{ sMin, sMax, uMax string }, 6) - for _, bits := range []int{8, 16, 32, 64, 128, 256} { - // big.Int because Int128/Int256 exceed every native width; rendered - // once to decimal strings so range checks are compareCanonicalDecimals - // calls. For bits=8 the three bounds are −128, 127, and 255. - pow := new(big.Int).Lsh(big.NewInt(1), uint(bits-1)) // 2^(bits−1) - sMin := new(big.Int).Neg(pow) // signed min: −2^(bits−1) - sMax := new(big.Int).Sub(pow, big.NewInt(1)) // signed max: 2^(bits−1) − 1 - uMax := new(big.Int).Sub(new(big.Int).Lsh(pow, 1), big.NewInt(1)) // unsigned max: 2^bits − 1 - m[bits] = struct{ sMin, sMax, uMax string }{sMin.String(), sMax.String(), uMax.String()} - } - return m -}() - -// integerInRange reports whether a canonical integer form lies within the -// column's width. An unknown width refuses — fail closed, never a guessed -// range. Integers alone need this explicit bounds table because their -// comparison is digit-string arithmetic with no inherent width; the float -// family's range gate is narrowFloat's ParseFloat-overflow refusal, and the -// decimal family's is decimalInPrecision. -func (n NumericSpec) integerInRange(c string) bool { - b, ok := intBounds[n.Bits] - if !ok { - return false - } - if n.Unsigned { - return compareCanonicalDecimals(c, "0") >= 0 && compareCanonicalDecimals(c, b.uMax) <= 0 - } - return compareCanonicalDecimals(c, b.sMin) >= 0 && compareCanonicalDecimals(c, b.sMax) <= 0 -} - -// decimalInPrecision reports whether a canonical form's integer digits fit the -// column's Precision−Scale budget. A payload past it is never storable -// (DECIMAL_OVERFLOW rejects the insert); a constant past it was measured to -// promote and compare mathematically on the query path — but the integer -// widths' wrap behavior (intBounds) shows the same class is not reliably -// modelable across pairs, so the refusal keeps one rule for every family at an -// availability-only cost. A lone "0" integer part spends no digits -// (Decimal(2,2) legally stores 0.99). -func (n NumericSpec) decimalInPrecision(c string) bool { - if n.Precision < 1 || n.Scale < 0 || n.Scale > n.Precision { - // No coherent model: refuse, fail closed. The Scale bounds also protect - // truncateScale's slicing — a hand-built spec must degrade to refusal, - // never a panic on the fan-out goroutine. - return false - } - intPart, _, _ := strings.Cut(strings.TrimPrefix(c, "-"), ".") - digits := len(intPart) - if intPart == "0" { - digits = 0 - } - return digits <= n.Precision-n.Scale -} - -// compare orders two canonical decimal operands (numericCanonical / -// CanonicalNumericLiteral output) in the column's storage domain, ok=false when -// the model refuses the pair. Narrowing BOTH sides is what ClickHouse itself -// does — it narrows the payload at insert and the bound constant at compare -// (verified: Float32 stores 16777217 as 16777216 and `= '16777217'` still -// matches; Decimal(10,2) stores 1.005 as 1.00 and `= '1.005'` still matches) — -// so a threshold filter can no longer admit an event whose stored row lands on -// the other side of the comparison (the ordering fail-open raised on #381). -func (n NumericSpec) compare(a, b string) (int, bool) { - switch n.Family { - case NumericInteger: - // A fractional CONSTANT against an integer column is a per-query type - // error on the SQL path (the role reads no rows, loudly); a fractional - // PAYLOAD was never storable in the column. Refuse both — fail closed. - if strings.Contains(a, ".") || strings.Contains(b, ".") { - return 0, false - } - // Same rule for the column's range: ClickHouse's reading of an - // out-of-range constant varies by pair (error, mathematical promotion, - // or a width-boundary wrap that compares against a DIFFERENT value - // than written — see intBounds), and an out-of-range payload was never - // storable. Refuse both sides rather than model any one behavior. - if !n.integerInRange(a) || !n.integerInRange(b) { - return 0, false - } - return compareCanonicalDecimals(a, b), true - case NumericFloat: - fa, ok := narrowFloat(a, n.Bits) - if !ok { - return 0, false - } - fb, ok := narrowFloat(b, n.Bits) - if !ok { - return 0, false - } - // Both operands are now exact values of the column's float domain, so - // direct comparison IS the domain comparison — no ties left to break. - switch { - case fa < fb: - return -1, true - case fa > fb: - return 1, true - default: - return 0, true - } - case NumericDecimal: - // Precision is the range gate of the decimal family: a payload with - // integer digits past Precision−Scale is never storable (the insert is - // rejected), while a constant past it was measured to PROMOTE on the - // query path and compare mathematically — the refusal there is an - // accepted availability cost, taken because the integer widths' wrap - // behavior proves this class has no reliable single model (see - // decimalInPrecision — one story across the three sites). Scale - // truncation below cannot change integer digits, so gating - // pre-truncation is exact. - if !n.decimalInPrecision(a) || !n.decimalInPrecision(b) { - return 0, false - } - return compareCanonicalDecimals(truncateScale(a, n.Scale), truncateScale(b, n.Scale)), true - case NumericNone: - return 0, false // no storage model: refuse, fail closed - default: - return 0, false // future family nobody taught this switch: same refusal - } -} - -// narrowFloat converts a canonical decimal form to the column's float domain -// with a single correct rounding (ParseFloat at the exact bit width — never a -// float64 detour, whose double rounding can land Float32 values one ULP off). -// A magnitude the domain can't hold refuses the comparison (ParseFloat reports -// the overflow as an error — the float family's range gate): ClickHouse would -// store ±Inf there, and matching infinities is a verdict this evaluator can't -// prove cheaply, so the row is withheld — availability, never exposure. An -// unknown width refuses too: ParseFloat silently treats any other bitSize as -// 64, which would compare a Float32 column in the wrong (wider) domain — the -// fail-open direction — instead of the zero-value-refuses contract the -// integer and decimal families keep. -func narrowFloat(canonical string, bits int) (float64, bool) { - if bits != 32 && bits != 64 { - return 0, false - } - f, err := strconv.ParseFloat(canonical, bits) - if err != nil { - return 0, false - } - return f, true -} - -// truncateScale narrows a canonical decimal form to scale fractional digits, -// truncating toward zero — ClickHouse's Decimal cast (1.005, 1.006 and 1.009 -// all store as 1.00 in a Decimal(10,2); rounding would predict 1.01). The -// result is re-canonicalized (trailing zeros trimmed, bare "-0" folded) so it -// stays valid compareCanonicalDecimals input. -func truncateScale(canonical string, scale int) string { - intPart, frac, hasFrac := strings.Cut(canonical, ".") - if !hasFrac { - return canonical - } - if len(frac) > scale { - frac = frac[:scale] - } - for len(frac) > 0 && frac[len(frac)-1] == '0' { - frac = frac[:len(frac)-1] - } - if len(frac) > 0 { - return intPart + "." + frac - } - if intPart == "-0" { - return "0" - } - return intPart -} - -// compareCanonicalDecimals orders two canonical decimal forms (CanonicalScalar -// output) as numbers, by digit-string arithmetic alone — the comparison twin of -// canonicalDecimal, sharing its invariants: an optional leading '-' (never on -// zero), no leading integer zeros except a lone "0", no trailing fraction -// zeros, no exponent. Those invariants are what make the string operations -// sound: with no leading zeros a longer integer part IS the larger magnitude, -// and with no trailing zeros a fraction that is a proper prefix of another IS -// the smaller. Never a float round-trip, so 64-bit-plus IDs order exactly. -func compareCanonicalDecimals(a, b string) int { - if a == b { - return 0 - } - na, nb := strings.HasPrefix(a, "-"), strings.HasPrefix(b, "-") - switch { - case na && !nb: - return -1 - case !na && nb: - return 1 - case na && nb: - return -compareCanonicalMagnitudes(a[1:], b[1:]) - } - return compareCanonicalMagnitudes(a, b) -} - -// compareCanonicalMagnitudes orders two unsigned canonical forms. -func compareCanonicalMagnitudes(a, b string) int { - ai, af, _ := strings.Cut(a, ".") - bi, bf, _ := strings.Cut(b, ".") - if len(ai) != len(bi) { - if len(ai) < len(bi) { - return -1 - } - return 1 - } - if c := strings.Compare(ai, bi); c != 0 { - return c - } - return strings.Compare(af, bf) -} diff --git a/internal/policy/policy.go b/internal/policy/policy.go index 61027a39..046a9304 100644 --- a/internal/policy/policy.go +++ b/internal/policy/policy.go @@ -115,13 +115,10 @@ type ResolvedPermissions struct { type ResolvedSelect struct { AllowColumns []string DenyColumns []string - WhereClause string - WhereParams []any - // rowFilter is the same row-level-security predicate as WhereClause/WhereParams, - // kept in resolved form so the stream path can evaluate it against the row - // (Predicates, RowVisible) while the query path renders it to SQL (WhereSQL). - // Both derive from one resolvePredicates call in Evaluate, so the two read - // surfaces can't drift (#457). + // rowFilter is the row-level-security predicate in resolved form: the query + // path renders it to SQL (WhereSQL) and the stream path hands it to the type + // layer (Predicates). Both derive from one resolvePredicates call in + // Evaluate, so the two read surfaces can't drift (#457). rowFilter []Predicate AllowedAggregations []string DeniedAggregations []string @@ -268,17 +265,9 @@ func evaluateSelect(perms *SelectPermissions, claims map[string]any) *ResolvedPe return &ResolvedPermissions{Allowed: false} } } - // Resolve the row-filter once into predicates, then render both read surfaces - // from that single source so they can't drift: the query path binds them into - // a SQL WHERE here; the stream path evaluates the same predicates in memory - // (ResolvedPermissions.RowVisible). - preds := resolvePredicates(perms.Filter, claims) - resolved.Select.rowFilter = preds - clauses, params := predicatesToSQL(preds, nil) - if len(clauses) > 0 { - resolved.Select.WhereClause = strings.Join(clauses, " AND ") - resolved.Select.WhereParams = params - } + // Resolve the row-filter once into predicates; both read surfaces render + // from that single source (WhereSQL, Predicates) so they can't drift. + resolved.Select.rowFilter = resolvePredicates(perms.Filter, claims) } return resolved @@ -321,15 +310,8 @@ func evaluateInsert(perms *InsertPermissions, claims map[string]any) *ResolvedPe case f.Eq != nil: // Deliberate asymmetry with the read path: an unresolvable check claim // still resolves to "" and is auto-injected as the required value (#463). - // A placeholder-free value is marked LiteralValue so the check - // comparison can accept its numeric reading; a claim-derived value - // stays a plain string and keeps strict canonical equality. v, _ := resolveTemplate(*f.Eq, claims) - if !claimTemplateRe.MatchString(*f.Eq) { - resolved.Insert.CheckClauses[col] = LiteralValue(v) - } else { - resolved.Insert.CheckClauses[col] = v - } + resolved.Insert.CheckClauses[col] = v case f.In != nil: // A []any value marks a set-membership check (vs a scalar required // value); ingest enforces "inserted value must be one of these". @@ -341,6 +323,29 @@ func evaluateInsert(perms *InsertPermissions, claims map[string]any) *ResolvedPe return resolved } +// HasRowFilter reports whether this role/table entry carries a row-level-security +// predicate. The stream fan-out uses it to decide whether an event can be projected +// once for a whole role bucket (no filter) or must be checked per subscriber against +// that subscriber's claims (filter present). A nil receiver (no policy applies) has +// no filter. +// +// It answers YES for a denied grant and for one whose read side was never +// resolved, neither of which has a predicate to speak of. That is deliberate: +// this is the GATE in front of the per-subscriber check, and a "no filter" +// answer sends the caller down the deliver-to-the-whole-bucket fast path where +// that check is never consulted. Saying yes forces the per-subscriber path, +// where the check denies. Same shape and same reason as RestrictsColumns, which +// guards the builder's SELECT * expansion. +func (rp *ResolvedPermissions) HasRowFilter() bool { + if rp == nil { + return false + } + if !rp.Allowed || rp.Select == nil { + return true + } + return len(rp.Select.rowFilter) > 0 +} + // Predicate is one row-filter comparison with its claim templates already // resolved to concrete string values — the shared, render-agnostic form the query // path turns into SQL (predicatesToSQL) and the stream path evaluates against the @@ -488,9 +493,7 @@ func toStrings(vals []any) []string { // with no placeholders — including a literal "" — is always ok, and binds // exactly as written: canonicalizing a numeric-spelled literal here would // silently move read filters on String columns (`_neq: "1.0"` on a version -// column is a different predicate than `_neq: "1"`); the insert-check -// comparison instead accepts a literal's numeric reading at compare time -// (CanonicalNumericLiteral). +// column is a different predicate than `_neq: "1"`). func resolveTemplate(tmpl string, claims map[string]any) (string, bool) { ok := true resolved := claimTemplateRe.ReplaceAllStringFunc(tmpl, func(match string) string { @@ -866,7 +869,7 @@ func validateSelectPerms(table, role string, perms *SelectPermissions) error { } // An entry naming no operator resolves to no predicate, so the row-level // restriction the author declared would silently not apply — Evaluate would - // answer HasRowFilter() false and RowVisible true for every row. Refuse it + // answer HasRowFilter() false and read every row. Refuse it // here; the resolver's matching deny is defense-in-depth. if !f.hasOperator() { return fmt.Errorf("table %q, op %q, role %q: filter column %q sets no operator — use _eq, _neq, _gt, _lt, or _in", table, op, role, col) diff --git a/internal/policy/policy_test.go b/internal/policy/policy_test.go index 0362a56d..2f0d851d 100644 --- a/internal/policy/policy_test.go +++ b/internal/policy/policy_test.go @@ -127,9 +127,10 @@ func TestEvaluate_FilterWithClaimTemplate(t *testing.T) { claims := map[string]any{"org_id": "org-123"} perms := Evaluate(p, "user", "clicks", "select", claims) assert.True(t, perms.Allowed) - assert.Contains(t, perms.Select.WhereClause, "`org_id` = ?") - require.Len(t, perms.Select.WhereParams, 1) - assert.Equal(t, "org-123", perms.Select.WhereParams[0]) + where, whereParams := perms.Select.WhereSQL(nil) + assert.Contains(t, where, "`org_id` = ?") + require.Len(t, whereParams, 1) + assert.Equal(t, "org-123", whereParams[0]) } func TestEvaluate_CheckClauses(t *testing.T) { @@ -151,28 +152,6 @@ func TestEvaluate_CheckClauses(t *testing.T) { assert.Equal(t, "org-456", perms.Insert.CheckClauses["org_id"]) } -// TestEvaluate_CheckClauses_StaticLiteralTyped: a placeholder-free check -// value is wrapped as LiteralValue — the marker that lets the ingest -// comparison accept its numeric reading — while a claim-derived value (above) -// stays a plain string, so a string-typed claim can never gain that reading. -func TestEvaluate_CheckClauses_StaticLiteralTyped(t *testing.T) { - t.Parallel() - eqVal := "1.0" - p := &Policy{ - Tables: map[string]TablePolicy{ - "clicks": { - "user": {Insert: &InsertPermissions{Check: map[string]Filter{ - "count": {Eq: &eqVal}, - }}}, - }, - }, - } - perms := Evaluate(p, "user", "clicks", "insert", map[string]any{}) - assert.True(t, perms.Allowed) - require.Contains(t, perms.Insert.CheckClauses, "count") - assert.Equal(t, LiteralValue("1.0"), perms.Insert.CheckClauses["count"]) -} - func TestEvaluate_AggregationLimits(t *testing.T) { t.Parallel() p := &Policy{ @@ -471,8 +450,7 @@ func TestResolveTemplate(t *testing.T) { // A static literal binds exactly as written even when it spells a JSON // number: canonicalizing it here would move read filters on String // columns (`_neq: "1.0"` on a version column would stop excluding rows - // storing "1.0"). The insert-check comparison accepts the numeric - // reading at compare time instead (CanonicalNumericLiteral). + // storing "1.0"). The insert check judges the literal in ClickHouse. {"numeric-spelled literal binds as written", "1.0", "1.0", true}, {"exponent-spelled literal binds as written", "1e400", "1e400", true}, } @@ -578,41 +556,6 @@ func TestCanonicalScalar(t *testing.T) { } } -// TestCanonicalNumericLiteral pins the numeric reading of a policy-authored -// check literal: only spellings JSON itself can produce canonicalize — the -// json.Valid gate rejects big.Int-acceptable forms like "+5" and "007" that -// no decoded claim or payload value ever carries, so the check comparison's -// second reading can't accept a spelling the first side can't produce. -func TestCanonicalNumericLiteral(t *testing.T) { - t.Parallel() - tests := []struct { - name string - lit string - want string - ok bool - }{ - {"float spelling of an integer", "1.0", "1", true}, - {"exponent spelling", "25e-4", "0.0025", true}, - {"negative fraction", "-2.50", "-2.5", true}, - {"integer passes through", "7", "7", true}, - {"leading plus is not JSON", "+5", "", false}, - {"leading zero is not JSON", "007", "", false}, - {"whitespace-padded number is not a bare literal", " 5", "", false}, - {"non-numeric literal", "org-123", "", false}, - {"boolean literal is valid JSON but not a number", "true", "", false}, - {"empty literal", "", "", false}, - {"no canonical form past the bound", "1e400", "", false}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - got, ok := CanonicalNumericLiteral(tt.lit) - assert.Equal(t, tt.want, got) - assert.Equal(t, tt.ok, ok) - }) - } -} - func TestValidate(t *testing.T) { t.Parallel() tests := []struct { @@ -1029,8 +972,9 @@ func TestEvaluate_FilterUnresolvableClaim_FailsClosed(t *testing.T) { }} perms := Evaluate(p, "user", "clicks", "select", map[string]any{"role": "user"}) require.True(t, perms.Allowed) - assert.Equal(t, "1 = 0", perms.Select.WhereClause) - assert.Empty(t, perms.Select.WhereParams) + where, whereParams := perms.Select.WhereSQL(nil) + assert.Equal(t, "1 = 0", where) + assert.Empty(t, whereParams) } // TestValidate_RejectsBindUnsafeFilterColumn: a policy whose row-filter column @@ -1181,7 +1125,7 @@ func TestResolveFilters_InNumericElements_BindCanonically(t *testing.T) { // exact digits; a float64 below 2^53 binds positionally (never the "1e+06" // spelling ClickHouse integer columns reject); a float64 at or past 2^53 lost // its digits at decode, so the predicate renders `1 = 0` — matching no rows, the -// same verdict RowVisible reaches in memory — alone or as one _in element. +// same verdict the type layer reaches — alone or as one _in element. func TestResolveFilters_NumericClaimBinding(t *testing.T) { t.Parallel() tmpl := "{{ jwt.tenant }}" @@ -1206,36 +1150,6 @@ func TestResolveFilters_NumericClaimBinding(t *testing.T) { assert.Empty(t, params) } -// TestCompareCanonicalDecimals pins the digit-string ordering over canonical -// forms — the comparison twin of canonicalDecimal, exact at any width, never a -// float round-trip. Each pair is asserted in both directions. -func TestCompareCanonicalDecimals(t *testing.T) { - t.Parallel() - tests := []struct { - a, b string - want int - }{ - {"0", "0", 0}, - {"1", "2", -1}, - {"9", "100", -1}, - {"-1", "1", -1}, - {"-2", "-1", -1}, - {"-100", "-9", -1}, - {"1.5", "1.5", 0}, - {"1.05", "1.5", -1}, - {"0.5", "0.55", -1}, - {"2", "2.5", -1}, - {"-1.5", "-1", -1}, - {"0.0025", "0.003", -1}, - {"12345678901234567890", "12345678901234567891", -1}, - {"9007199254740992", "9007199254740993", -1}, - } - for _, tt := range tests { - assert.Equal(t, tt.want, compareCanonicalDecimals(tt.a, tt.b), "%s vs %s", tt.a, tt.b) - assert.Equal(t, -tt.want, compareCanonicalDecimals(tt.b, tt.a), "%s vs %s reversed", tt.b, tt.a) - } -} - // TestResolveFilters_InEmptyClaim_FailsClosed: an empty set makes the predicate // match no rows (a constant-false predicate) rather than widen to all rows — the // fail-closed direction. `IN ()` is invalid SQL. Two distinct branches of @@ -1263,7 +1177,7 @@ func TestResolveFilters_InEmptyClaim_FailsClosed(t *testing.T) { } // TestEvaluate_FilterInClause: end-to-end through Evaluate, an _in filter lands -// in the role's WhereClause/WhereParams (exercising the bind-safe guard + IN +// in the role's WhereSQL (exercising the bind-safe guard + IN // assembly), not just the resolveFilters unit. func TestEvaluate_FilterInClause(t *testing.T) { t.Parallel() @@ -1274,8 +1188,9 @@ func TestEvaluate_FilterInClause(t *testing.T) { claims := map[string]any{"app_metadata": map[string]any{"tenant_ids": []any{"t1", "t2"}}} perms := Evaluate(p, "user", "clicks", "select", claims) require.True(t, perms.Allowed) - assert.Contains(t, perms.Select.WhereClause, "`tenant_id` IN (?,?)") - assert.Equal(t, []any{"t1", "t2"}, perms.Select.WhereParams) + where, whereParams := perms.Select.WhereSQL(nil) + assert.Contains(t, where, "`tenant_id` IN (?,?)") + assert.Equal(t, []any{"t1", "t2"}, whereParams) } // TestEvaluate_CheckInResolvesToSet: an _in check resolves to a []any set in @@ -1550,14 +1465,10 @@ func TestEvaluate_UnresolvedSideFailsClosed(t *testing.T) { assert.False(t, ins.IsAggregationAllowed("count")) assert.True(t, ins.HasRowFilter(), "the GATE must not report 'no filter' — that sends the caller down the "+ - "whole-bucket fast path where RowVisible is never consulted") - assert.False(t, ins.RowVisible(map[string]any{"tenant_id": "acme"}, nil), - "an unresolved read side must not admit every row") + "whole-bucket fast path") // The select-resolved grant still evaluates its row filter normally. assert.True(t, sel.HasRowFilter()) - assert.True(t, sel.RowVisible(map[string]any{"tenant_id": "acme"}, nil)) - assert.False(t, sel.RowVisible(map[string]any{"tenant_id": "globex"}, nil)) } // TestHandBuiltPermissions_PresentSidesKeepPlainReading: a value assembled by @@ -1571,7 +1482,6 @@ func TestHandBuiltPermissions_PresentSidesKeepPlainReading(t *testing.T) { assert.True(t, rp.IsColumnAllowed("anything", true)) assert.True(t, rp.IsAggregationAllowed("count")) assert.False(t, rp.RestrictsColumns()) - assert.True(t, rp.RowVisible(map[string]any{"a": 1}, nil)) // An EMPTY insert side has no checks and is resolved — not the same answer as // a nil one below. A slip to `rp.Insert != nil && len(...) > 0` would break // exactly here. @@ -1593,10 +1503,9 @@ func TestHandBuiltPermissions_NilSideDenies(t *testing.T) { assert.False(t, insertOnly.IsAggregationAllowed("count")) assert.True(t, insertOnly.RestrictsColumns(), "an unresolved read side restricts everything") // HasRowFilter says YES on an unresolved read side on purpose: it routes the - // hub onto the per-subscriber path where RowVisible denies, instead of the - // no-filter fast path that never consults RowVisible at all. + // hub onto the per-subscriber path, which denies, instead of the no-filter + // fast path. assert.True(t, insertOnly.HasRowFilter(), "must not take the no-filter fast path") - assert.False(t, insertOnly.RowVisible(map[string]any{"a": 1}, nil), "and the per-row check denies") assert.Empty(t, insertOnly.AllowedProjection([]string{"a", "b"})) _, insertChecksOK := insertOnly.CheckClauses() @@ -1674,7 +1583,7 @@ func TestEvaluate_OperatorLessFilterAndCheckDenyFailClosed(t *testing.T) { // `"tenant_id": {}` survives a strict decode — every Filter operator is // omitempty — and then matches no case in either resolver, so the declared // restriction resolves to nothing. Before this was refused, the policy below - // validated clean and RowVisible answered true for every tenant. + // validated clean and every tenant could read. sel := &Policy{Tables: map[string]TablePolicy{ "clicks": {"viewer": {Select: &SelectPermissions{ AllowColumns: []string{"*"}, @@ -1700,7 +1609,6 @@ func TestEvaluate_OperatorLessFilterAndCheckDenyFailClosed(t *testing.T) { selPerms := Evaluate(sel, "viewer", "clicks", "select", nil) assert.False(t, selPerms.Allowed, "an operator-less filter must deny, not read as unrestricted") assert.True(t, selPerms.HasRowFilter(), "a denied grant gates every row") - assert.False(t, selPerms.RowVisible(map[string]any{"tenant_id": "someone-else"}, nil)) insPerms := Evaluate(ins, "writer", "clicks", "insert", nil) assert.False(t, insPerms.Allowed, "an operator-less check must deny, not drop the rule") @@ -1757,7 +1665,7 @@ func TestPredicates_UnresolvableClaim_NoValuesOnBothPaths(t *testing.T) { } } -// TestPredicates_IsTheSameResolutionAsTheWhereClause: the two read surfaces are +// TestPredicates_IsTheSameResolutionAsTheWhereSQL: the two read surfaces are // rendered from ONE resolvePredicates call, so the predicates handed to the // stream carry exactly the values bound into the query's WHERE — in the same // order. A second resolution, even of the same policy, is what #457 was about. @@ -1773,8 +1681,6 @@ func TestPredicates_IsTheSameResolutionAsTheWhereClause(t *testing.T) { clause, params := perms.Select.WhereSQL(nil) assert.Equal(t, "`tenant_id` = ?", clause) assert.Equal(t, []any{"acme"}, params) - assert.Equal(t, perms.Select.WhereClause, clause, "WhereSQL(nil) is the WhereClause rendering") - assert.Equal(t, perms.Select.WhereParams, params) preds, ok := perms.Predicates() require.True(t, ok) diff --git a/internal/policy/rowfilter.go b/internal/policy/rowfilter.go deleted file mode 100644 index d4a9b087..00000000 --- a/internal/policy/rowfilter.go +++ /dev/null @@ -1,294 +0,0 @@ -package policy - -import ( - "encoding/json" - "strings" - "time" -) - -// HasRowFilter reports whether this role/table entry carries a row-level-security -// predicate. The stream fan-out uses it to decide whether an event can be projected -// once for a whole role bucket (no filter) or must be checked per subscriber against -// that subscriber's claims (filter present). A nil receiver (no policy applies) has -// no filter. -// -// It answers YES for a denied grant and for one whose read side was never -// resolved, neither of which has a predicate to speak of. That is deliberate: -// this is the GATE in front of RowVisible, and a "no filter" answer sends the -// caller down the deliver-to-the-whole-bucket fast path where RowVisible is -// never consulted. Saying yes forces the per-subscriber path, where RowVisible -// denies. Same shape and same reason as RestrictsColumns, which guards the -// builder's SELECT * expansion. -func (p *ResolvedPermissions) HasRowFilter() bool { - if p == nil { - return false - } - if !p.Allowed || p.Select == nil { - return true - } - return len(p.Select.rowFilter) > 0 -} - -// maxTimeOperandChars is the same O(1) pre-gate for timestamp operands: the -// ingest grammar's longest accepted spelling (RFC 3339 with nanoseconds and a -// numeric offset) is 35 bytes, so 64 is generous slack — and the parser scans -// its input, which without the gate a megabyte "timestamp" would make a -// per-subscriber-per-event cost. -const maxTimeOperandChars = 64 - -// ColumnKind classifies a column's ClickHouse type for the in-memory row-filter -// comparison. The zero value is ColumnOpaque, so a nil map, a column absent from -// the map, and a column the schema doesn't know all land on the most conservative -// class — the three "no type knowledge" states are indistinguishable and equally -// closed, never a silent downgrade to a laxer comparison. -type ColumnKind uint8 - -const ( - // ColumnOpaque: no usable type knowledge (no schema, unknown column) or a type - // whose text rendering is not canonical — UUID (case), Enum (name vs number), - // Bool (true vs 1), Date/Date32 (producer spelling), IPv4/IPv6, … For these only - // byte-equality is trustworthy: identical strings parse to identical ClickHouse - // values, but differing strings prove nothing. So = and in admit exactly the - // event's own rendering, while !=, > and < fail closed (the row is withheld). - ColumnOpaque ColumnKind = iota - // ColumnNumeric (Int*/UInt*/Float*/Decimal*): both operands render to exact - // canonical decimal form through the claim side's #457 machinery, then - // compare in the column's STORAGE domain (ColumnSpec.Numeric): integers at - // any width exactly, floats after IEEE narrowing to the column's bit width, - // decimals after truncation to the column's scale — the same narrowing - // ClickHouse applies to the stored value and the bound constant, so stream - // and query verdicts agree even on narrowing columns. - ColumnNumeric - // ColumnText (String, incl. Nullable/LowCardinality): byte comparison is - // ClickHouse comparison — equality AND lexicographic order — so every operator - // is exact. FixedString is NOT ColumnText (zero-padded storage). - ColumnText - // ColumnTime (DateTime/DateTime64): operands parse as instants through the - // caller-supplied ColumnSpec.ParseTime — the same grammar, zone rule, and - // range guard ingest canonicalization applies — and compare chronologically, - // so every operator is exact across spellings: a zone-less filter constant - // matches the canonicalized RFC 3339 payload denoting the same instant. A - // side that can't be read as a provable instant fails closed. - ColumnTime -) - -// ColumnSpec is one column's comparison contract for the in-memory row filter: -// the ColumnKind classification plus the kind's parameters — ColumnTime's -// instant parser, ColumnNumeric's storage model. The zero value is ColumnOpaque -// with neither, so a nil map, an absent column, and an unknown type all land on -// the most conservative class — never a silent downgrade to a laxer comparison. -type ColumnSpec struct { - Kind ColumnKind - // ParseTime converts one rendering of this timestamp column's value — an - // ingested payload value (string / json.Number / float64) or a resolved - // filter constant (always a string) — to the instant ClickHouse would store, - // truncated to the column's precision. ok=false (unparseable, or outside the - // column type's range, which insert-time saturation would move) fails the - // comparison closed. Set iff Kind is ColumnTime; the stream supplies it from - // the schema registry (discovery's Column.TimeParser) so the filter and - // ingest canonicalization can never disagree on the grammar. - ParseTime func(v any) (t time.Time, ok bool) - // Numeric is the column's storage model, set iff Kind is ColumnNumeric - // (from discovery.NumericStorageOf via the stream's columnSpecs). Its zero - // value refuses every comparison, so a ColumnNumeric spec built without a - // model fails closed rather than comparing under the wrong semantics. - Numeric NumericSpec -} - -// RowVisible reports whether row satisfies every resolved row-filter predicate — the -// in-memory twin of the query path's WHERE clause, evaluated against a decoded event -// so the stream applies the same row-level security the query path does. Predicates -// are ANDed; the query path joins them with AND too. -// -// cols maps column name → ColumnSpec, supplied by the caller from the table -// schema (see stream.Hub's columnSpecs). Numeric columns compare numerically in -// the column's storage domain (9 < 100, as ClickHouse would; Float/Decimal -// operands narrowed the way insert and constant binding narrow them), String -// columns compare bytewise (exactly ClickHouse's String collation), -// DateTime/DateTime64 columns compare as instants (both operands parsed through -// the spec's ParseTime, the same grammar ingest canonicalizes with), and -// everything else — including every column when no schema is available — admits -// only byte-equality (= / in) and fails !=, > and < closed: the evaluator cannot -// mirror ClickHouse's per-type coercion, and text comparison there could admit rows -// the query path excludes ("9" > "100" as text, an uppercase UUID under !=). Every -// ambiguous or uncomparable case fails closed — the row is hidden, never leaked — -// so the boundary costs availability, not confidentiality. -// -// That guarantee is about the INGESTED PAYLOAD value, which is what the stream -// evaluates; the query path evaluates the stored row. Storage-domain narrowing -// keeps the two verdicts aligned for values ClickHouse stores; the residual -// asymmetry is an event whose INSERT later fails entirely (out-of-range value, -// batch error → DLQ): it was already streamed to whoever the filter admitted, -// and the row never becomes queryable. Documented in access-control.mdx's -// enforcement caution. -// -// A nil receiver (no policy applies) makes every row visible. -func (p *ResolvedPermissions) RowVisible(row map[string]any, cols map[string]ColumnSpec) bool { - if p == nil { - return true - } - // A denied role sees no rows — fail closed, mirroring the !Allowed guard on - // IsColumnAllowed, so a denied receiver never reads as "no filter ⇒ all visible". - // A grant resolved for INSERT is refused here for the same reason: its empty - // read side would otherwise read as "no filter", admitting every row. - if !p.Allowed || p.Select == nil { - return false - } - for _, pred := range p.Select.rowFilter { - if !pred.matches(row, cols[pred.Column]) { - return false - } - } - return true -} - -// matches evaluates one predicate against the row, failing closed (false) whenever -// the value is absent or can't be compared as required. -func (pred Predicate) matches(row map[string]any, spec ColumnSpec) bool { - // No values ⇒ matches nothing: an empty/unresolvable "in" set, or a scalar - // whose constant was unrenderable — the in-memory twin of the `1 = 0` - // predicatesToSQL emits for the same cases. - if len(pred.Values) == 0 { - return false - } - raw, ok := row[pred.Column] - if !ok { - return false // column not in the event ⇒ can't prove the row is allowed - } - switch pred.Op { - case "=": - c, ok := compareScalar(raw, pred.Values[0], spec) - return ok && c == 0 - case "!=": - c, ok := compareScalar(raw, pred.Values[0], spec) - return ok && c != 0 - case ">": - c, ok := compareScalar(raw, pred.Values[0], spec) - return ok && c > 0 - case "<": - c, ok := compareScalar(raw, pred.Values[0], spec) - return ok && c < 0 - case "in": - for _, v := range pred.Values { - if c, ok := compareScalar(raw, v, spec); ok && c == 0 { - return true - } - } - return false - default: - return false - } -} - -// compareScalar compares an event value against a resolved filter value, returning -// -1/0/+1 and ok=false when the comparison can't be made: a non-scalar event value, -// a numeric comparison whose operands don't parse as numbers, or two unequal values -// of a ColumnOpaque column (where inequality and order are unprovable). Every -// ok=false fails the enclosing predicate closed — for != that is what keeps a mere -// representation difference (an uppercase UUID, 1 for a Bool true) from being -// mistaken for a real inequality and admitting a row the query path excludes. -// filterVal is the raw resolved spelling — what byte-comparison arms compare -// and predicatesToSQL binds; the numeric arm derives its canonical reading per -// comparison, behind an O(1) length gate (see maxNumericOperandChars — folding -// the re-derivation into a memoized resolution is part of #435's scope). -func compareScalar(rowVal any, filterVal string, spec ColumnSpec) (int, bool) { - switch spec.Kind { - case ColumnTime: - // The RAW value goes to the parser — a payload timestamp may legitimately - // be a number (Unix seconds/ticks), which the spec's parser reads the same - // way ingest does; scalarString's rendering would be a detour. Both sides - // must parse; either failing (or a spec missing its parser) refuses the - // comparison, fail closed. - if spec.ParseTime == nil { - return 0, false - } - // O(1) length gate before the parser scans either operand (see - // maxTimeOperandChars). The payload arrives as a string OR a - // json.Number (a Unix-epoch timestamp) — both are client-controlled and - // must be gated; a bare-number payload that skipped this would leave the - // parser's digit scan (and its %q error rendering) unbounded on the - // fan-out path. Verdict-preserving for numbers (>19 digits never parsed) - // and, for strings, refuses only past 64 bytes, where Go's parser would - // have truncated sub-second digits rather than matched a real instant. - switch v := rowVal.(type) { - case string: - if len(v) > maxTimeOperandChars { - return 0, false - } - case json.Number: - if len(v) > maxTimeOperandChars { - return 0, false - } - } - if len(filterVal) > maxTimeOperandChars { - return 0, false - } - a, ok := spec.ParseTime(rowVal) - if !ok { - return 0, false - } - b, ok := spec.ParseTime(filterVal) - if !ok { - return 0, false - } - return a.Compare(b), true - case ColumnNumeric: - // Both operands route through the ONE canonical numeric gate the claim - // side already uses (#457's CanonicalScalar machinery): exact decimal - // form at any width, digit-bounded (the superlinear-parse guard lives - // there — CWE-400, the row operand is client-controlled), with "NaN", - // any Inf spelling, and every non-JSON-number rendering refused by the - // grammar rather than by ad-hoc checks. Then the comparison itself runs - // in the column's storage domain (NumericSpec.compare), narrowing both - // sides the way ClickHouse narrows the stored value and the constant. - if len(filterVal) > maxNumericOperandChars { - return 0, false - } - a, ok := numericCanonical(rowVal) - if !ok { - return 0, false - } - b, ok := CanonicalNumericLiteral(filterVal) - if !ok { - return 0, false - } - // Spelling fidelity, integer family only: predicatesToSQL binds the - // constant AS WRITTEN, and ClickHouse's integer cast in a WHERE rejects - // non-plain spellings ("1e3", "1.5", "007") with a per-query type - // error — the role reads no rows there, so comparing the canonical - // reading here would ADMIT rows SQL never returns. A constant that - // isn't its own canonical form refuses the comparison instead - // (withhold — matching SQL's nothing, just quietly; the one measured - // over-refusal is "-0", which ClickHouse accepts in a WHERE but the - // canonical fold rewrites to "0" — availability, never exposure). - // Claim-derived constants are canonical by construction and - // unaffected. Float AND Decimal casts accept every JSON-number - // spelling ('1e3' casts to Decimal as 1000 — verified), so neither - // family gets a gate. - if spec.Numeric.Family == NumericInteger && b != filterVal { - return 0, false - } - return spec.Numeric.compare(a, b) - case ColumnText: - s, ok := scalarString(rowVal) - if !ok { - return 0, false - } - return strings.Compare(s, filterVal), true - case ColumnOpaque: - // Byte-equality is the only relation provable without type knowledge - // (identical strings always parse to the same ClickHouse value); unequal - // bytes prove nothing, so the comparison is refused and the predicate - // fails closed. - s, ok := scalarString(rowVal) - if !ok { - return 0, false - } - if s == filterVal { - return 0, true - } - return 0, false - default: - return 0, false // unknown future kind: refuse to compare, fail closed - } -} diff --git a/internal/policy/rowfilter_test.go b/internal/policy/rowfilter_test.go deleted file mode 100644 index bda90f4d..00000000 --- a/internal/policy/rowfilter_test.go +++ /dev/null @@ -1,543 +0,0 @@ -package policy - -import ( - "encoding/json" - "strings" - "testing" - "time" - - "github.com/stretchr/testify/assert" -) - -// intNumeric is the Int64 numeric spec most tests compare under — exact within -// the width's range, the storage model of the default integer id column. Float -// and Decimal narrowing, other widths, and range gating have dedicated tests. -func intNumeric() ColumnSpec { - return ColumnSpec{Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericInteger, Bits: 64}} -} - -// evalRowFilter builds a one-role/one-table policy carrying filter and returns the -// permissions resolved against claims, so tests exercise the full -// resolvePredicates → RowVisible path the stream fan-out uses. -func evalRowFilter(t *testing.T, filter map[string]Filter, claims map[string]any) *ResolvedPermissions { - t.Helper() - p := &Policy{Tables: map[string]TablePolicy{ - "t": {"r": {Select: &SelectPermissions{Filter: filter}}}, - }} - return Evaluate(p, "r", "t", "select", claims) -} - -// TestRowVisible_EqualityScoping covers the canonical row-level-security shape — -// tenant_id = {{ jwt.tenant }} — including the fail-closed on a missing column that -// stops an event without the filtered field from slipping through. -func TestRowVisible_EqualityScoping(t *testing.T) { - t.Parallel() - perms := evalRowFilter(t, map[string]Filter{"tenant_id": {Eq: new("{{ jwt.tenant }}")}}, map[string]any{"tenant": "acme"}) - is := assert.New(t) - is.True(perms.HasRowFilter()) - is.True(perms.RowVisible(map[string]any{"tenant_id": "acme"}, nil)) - is.False(perms.RowVisible(map[string]any{"tenant_id": "globex"}, nil)) - is.False(perms.RowVisible(map[string]any{"other": "x"}, nil), "missing filtered column ⇒ fail closed") -} - -// TestRowVisible_InSet covers _in against an array claim (multi-tenant scoping) and -// the fail-closed empty-set case (absent claim never widens to all rows). -func TestRowVisible_InSet(t *testing.T) { - t.Parallel() - perms := evalRowFilter(t, map[string]Filter{"tenant_id": {In: new("{{ jwt.tenants }}")}}, map[string]any{"tenants": []any{"a", "b"}}) - assert.True(t, perms.RowVisible(map[string]any{"tenant_id": "b"}, nil)) - assert.False(t, perms.RowVisible(map[string]any{"tenant_id": "c"}, nil)) - - empty := evalRowFilter(t, map[string]Filter{"tenant_id": {In: new("{{ jwt.tenants }}")}}, nil) - assert.True(t, empty.HasRowFilter()) - assert.False(t, empty.RowVisible(map[string]any{"tenant_id": "a"}, nil), "absent claim ⇒ empty set ⇒ no row matches") -} - -// TestRowVisible_Neq: != admits a row only when inequality is PROVABLE — a numeric -// column comparing numerically or a String column comparing bytewise. On a column -// with no usable type (no schema, or a type like UUID/Bool whose text rendering -// isn't canonical) a byte difference may be pure representation — 'ABC-def' vs -// 'abc-def' for a UUID ClickHouse would treat as equal — so != fails closed rather -// than admit a row the query path excludes. -func TestRowVisible_Neq(t *testing.T) { - t.Parallel() - text := map[string]ColumnSpec{"status": {Kind: ColumnText}} - perms := evalRowFilter(t, map[string]Filter{"status": {Neq: new("deleted")}}, nil) - assert.True(t, perms.RowVisible(map[string]any{"status": "active"}, text), "String column: byte inequality is real inequality") - assert.False(t, perms.RowVisible(map[string]any{"status": "deleted"}, text)) - - assert.False(t, perms.RowVisible(map[string]any{"status": "active"}, nil), - "no schema: byte inequality proves nothing, fail closed") - - uuid := evalRowFilter(t, map[string]Filter{"device": {Neq: new("ABC-DEF")}}, nil) - assert.False(t, uuid.RowVisible(map[string]any{"device": "abc-def"}, map[string]ColumnSpec{"device": {Kind: ColumnOpaque}}), - "opaque column (e.g. UUID): a case difference is not proof of inequality — fail closed") - - num := evalRowFilter(t, map[string]Filter{"amount": {Neq: new("100")}}, nil) - kinds := map[string]ColumnSpec{"amount": intNumeric()} - assert.True(t, num.RowVisible(map[string]any{"amount": float64(250)}, kinds), "numeric column: 250 ≠ 100 is provable") - assert.False(t, num.RowVisible(map[string]any{"amount": float64(100)}, kinds)) -} - -// TestRowVisible_Ordering_SchemaInformed: ordering needs type knowledge. An event -// value of 9 is numerically LESS than 100 but lexicographically GREATER ("9" > -// "100") — so a numeric column compares numerically (matching ClickHouse), a String -// column compares bytewise (which IS ClickHouse's String order), and a column with -// no usable type fails closed: no schema, a column the schema doesn't know, and a -// non-numeric non-String type (Enum order follows enum values, not names; Date -// formats vary) must all withhold rather than fall back to a text comparison that -// can admit rows the query path excludes. -func TestRowVisible_Ordering_SchemaInformed(t *testing.T) { - t.Parallel() - perms := evalRowFilter(t, map[string]Filter{"amount": {Gt: new("100")}}, nil) - small := map[string]any{"amount": float64(9)} - big := map[string]any{"amount": float64(250)} - numeric := map[string]ColumnSpec{"amount": intNumeric()} - - assert.False(t, perms.RowVisible(small, numeric), "numeric: 9 is not > 100") - assert.True(t, perms.RowVisible(big, numeric)) - - assert.False(t, perms.RowVisible(small, nil), `no schema: fail closed — never the "9" > "100" text leak`) - assert.False(t, perms.RowVisible(big, nil), "no schema: fail closed even when the numbers would pass") - assert.False(t, perms.RowVisible(small, map[string]ColumnSpec{"other": intNumeric()}), - "column absent from a known schema: fail closed") - assert.False(t, perms.RowVisible(small, map[string]ColumnSpec{"amount": {Kind: ColumnOpaque}}), - "opaque type (Enum/Date/UUID/…): order is unprovable, fail closed") - - // String columns order bytewise in ClickHouse, so ordering there is exact. - page := evalRowFilter(t, map[string]Filter{"page": {Gt: new("/m")}}, nil) - text := map[string]ColumnSpec{"page": {Kind: ColumnText}} - assert.True(t, page.RowVisible(map[string]any{"page": "/z"}, text)) - assert.False(t, page.RowVisible(map[string]any{"page": "/a"}, text)) -} - -// TestRowVisible_NumericEquality_FloatFormatting: a JSON float64(100) equals the -// string filter value "100" under numeric comparison, so integer-valued numeric -// columns aren't tripped up by float formatting. -func TestRowVisible_NumericEquality_FloatFormatting(t *testing.T) { - t.Parallel() - perms := evalRowFilter(t, map[string]Filter{"amount": {Eq: new("100")}}, nil) - num := map[string]ColumnSpec{"amount": intNumeric()} - assert.True(t, perms.RowVisible(map[string]any{"amount": float64(100)}, num)) - assert.False(t, perms.RowVisible(map[string]any{"amount": float64(101)}, num)) -} - -// TestRowVisible_NaN_FailsClosed: strconv.ParseFloat accepts "NaN", and NaN's -// three-way comparison would otherwise read as "equal to everything" — a fail-open -// that delivers a row the query path's WHERE excludes. Either operand parsing to -// NaN must withhold the row instead. -func TestRowVisible_NaN_FailsClosed(t *testing.T) { - t.Parallel() - num := map[string]ColumnSpec{"amount": intNumeric()} - - byRow := evalRowFilter(t, map[string]Filter{"amount": {Eq: new("100")}}, nil) - assert.False(t, byRow.RowVisible(map[string]any{"amount": "NaN"}, num), "NaN row value must not equal 100") - - byClaim := evalRowFilter(t, map[string]Filter{"amount": {Eq: new("{{ jwt.cap }}")}}, map[string]any{"cap": "NaN"}) - assert.False(t, byClaim.RowVisible(map[string]any{"amount": float64(100)}, num), "NaN claim must not match any row") - - neq := evalRowFilter(t, map[string]Filter{"amount": {Neq: new("NaN")}}, nil) - assert.False(t, neq.RowVisible(map[string]any{"amount": float64(100)}, num), "NaN is uncomparable, so even != fails closed") -} - -// TestRowVisible_Inf_FailsClosed: ParseFloat also accepts "Inf"/"+Inf"/"-Inf"/ -// "Infinity" (any case), and an infinite bound would make _gt/_lt admit every -// finite row — a fail-open the query path can't reproduce (ClickHouse rejects -// binding an Inf-spelled string to an integer column). Either operand parsing -// to ±Inf must withhold the row instead. -func TestRowVisible_Inf_FailsClosed(t *testing.T) { - t.Parallel() - num := map[string]ColumnSpec{"amount": intNumeric()} - - byClaim := evalRowFilter(t, map[string]Filter{"amount": {Gt: new("{{ jwt.min }}")}}, map[string]any{"min": "-Inf"}) - assert.False(t, byClaim.RowVisible(map[string]any{"amount": float64(5)}, num), "-Inf lower bound must not admit finite rows") - - lt := evalRowFilter(t, map[string]Filter{"amount": {Lt: new("Infinity")}}, nil) - assert.False(t, lt.RowVisible(map[string]any{"amount": float64(5)}, num), "Infinity upper bound must not admit finite rows") - - byRow := evalRowFilter(t, map[string]Filter{"amount": {Gt: new("100")}}, nil) - assert.False(t, byRow.RowVisible(map[string]any{"amount": "+Inf"}, num), "Inf row value is uncomparable, fail closed") -} - -// TestRowVisible_NumericEquality_ExactBeyondFloat64: ingest accepts string-encoded -// numerics precisely so 64-bit IDs survive JS precision loss; equality must not -// collapse distinct IDs that round to the same float64 (adjacent values past 2^53). -func TestRowVisible_NumericEquality_ExactBeyondFloat64(t *testing.T) { - t.Parallel() - num := map[string]ColumnSpec{"id": intNumeric()} - - perms := evalRowFilter(t, map[string]Filter{"id": {Eq: new("9007199254740993")}}, nil) - assert.False(t, perms.RowVisible(map[string]any{"id": "9007199254740992"}, num), "float64-equal neighbors are not equal") - assert.True(t, perms.RowVisible(map[string]any{"id": "9007199254740993"}, num)) - assert.True(t, perms.RowVisible(map[string]any{"id": "9007199254740993.0"}, num), "same value in a different rendering still matches") - - // Bare JSON numbers reach RowVisible as json.Number (the stream decodes with - // UseNumber), so the same exactness holds without string-encoding: a lossy - // float64 decode would have collapsed these neighbors and delivered another - // tenant's row. - assert.False(t, perms.RowVisible(map[string]any{"id": json.Number("9007199254740992")}, num), "json.Number neighbor is not equal") - assert.True(t, perms.RowVisible(map[string]any{"id": json.Number("9007199254740993")}, num)) - - neq := evalRowFilter(t, map[string]Filter{"id": {Neq: new("9007199254740993")}}, nil) - assert.True(t, neq.RowVisible(map[string]any{"id": "9007199254740992"}, num), "the exact comparison keeps distinct IDs unequal for !=") -} - -// TestRowVisible_OverlongNumericOperand_FailsClosed: an over-long operand is -// refused by the O(1) length pre-gate (maxNumericOperandChars) before ANY scan -// of its bytes — the row operand is client-controlled and the comparison runs -// per subscriber per event on the fan-out goroutine, so even a linear -// json.Valid pass over it is a cost an attacker controls. The gate is -// verdict-preserving: anything longer already fails the canonical digit bound. -// Withheld, never read; real-width values unaffected. -func TestRowVisible_OverlongNumericOperand_FailsClosed(t *testing.T) { - t.Parallel() - num := map[string]ColumnSpec{"amount": intNumeric()} - perms := evalRowFilter(t, map[string]Filter{"amount": {Eq: new("100")}}, nil) - - long := "100." + strings.Repeat("0", 200_000) + "1" // far past the operand length gate - assert.False(t, perms.RowVisible(map[string]any{"amount": json.Number(long)}, num)) - assert.False(t, perms.RowVisible(map[string]any{"amount": long}, num), "string-encoded operand is bounded too") - assert.True(t, perms.RowVisible(map[string]any{"amount": json.Number("100.0")}, num), "real-width values still compare") -} - -// TestRowVisible_LossyFloatClaim_FailsClosed: a claim that arrives as a float64 -// at or past 2^53 has already collapsed onto its float neighbors — rendering -// digits for it could name another tenant (the fail-open caught in #381 review: -// the claim rendered as "1e+16" and matched the neighbor's rows). CanonicalScalar -// now refuses such a float64 outright, so the predicate matches NOTHING: not the -// neighbor the float equals, and not even the row whose exact ID the claim -// originally carried — availability, never confidentiality. -func TestRowVisible_LossyFloatClaim_FailsClosed(t *testing.T) { - t.Parallel() - num := map[string]ColumnSpec{"tenant_id": intNumeric()} - filter := map[string]Filter{"tenant_id": {Eq: new("{{ jwt.tenant }}")}} - - lossy := evalRowFilter(t, filter, map[string]any{"tenant": float64(10000000000000001)}) // already 1e16 - assert.False(t, lossy.RowVisible(map[string]any{"tenant_id": json.Number("10000000000000000")}, num), - "the float64-equal neighbor tenant's rows must not be delivered") - assert.False(t, lossy.RowVisible(map[string]any{"tenant_id": json.Number("10000000000000001")}, num), - "the original tenant's own rows are withheld too — the digits are unrecoverable") - - // The exact-digit path — a json.Number claim, which is what jwt.Parse yields - // since WithJSONNumber — scopes to precisely one tenant. - exact := evalRowFilter(t, filter, map[string]any{"tenant": json.Number("10000000000000001")}) - assert.True(t, exact.RowVisible(map[string]any{"tenant_id": json.Number("10000000000000001")}, num)) - assert.False(t, exact.RowVisible(map[string]any{"tenant_id": json.Number("10000000000000000")}, num)) -} - -// rfc3339Spec is a ColumnTime spec whose parser reads RFC 3339 strings — a -// hermetic stand-in for discovery's grammar, which is exercised in -// internal/discovery (Column.TimeParser) and end-to-end in the hub tests. -func rfc3339Spec() ColumnSpec { - return ColumnSpec{Kind: ColumnTime, ParseTime: func(v any) (time.Time, bool) { - s, ok := v.(string) - if !ok { - return time.Time{}, false - } - ts, err := time.Parse(time.RFC3339, s) - return ts, err == nil - }} -} - -// TestRowVisible_TimeColumn: DateTime/DateTime64 operands compare as instants, -// so equality holds across spellings of the same instant, ordering works (the -// time-window policy shape), and any operand the parser refuses — junk on either -// side, or a ColumnTime spec missing its parser — withholds the row. -func TestRowVisible_TimeColumn(t *testing.T) { - t.Parallel() - cols := map[string]ColumnSpec{"created_at": rfc3339Spec()} - - eq := evalRowFilter(t, map[string]Filter{"created_at": {Eq: new("2026-06-21T06:00:00+02:00")}}, nil) - assert.True(t, eq.RowVisible(map[string]any{"created_at": "2026-06-21T04:00:00Z"}, cols), - "different spelling, same instant ⇒ equal") - assert.False(t, eq.RowVisible(map[string]any{"created_at": "2026-06-21T04:00:01Z"}, cols)) - assert.False(t, eq.RowVisible(map[string]any{"created_at": "junk"}, cols), "unparseable payload withholds") - - gt := evalRowFilter(t, map[string]Filter{"created_at": {Gt: new("2026-06-21T00:00:00Z")}}, nil) - assert.True(t, gt.RowVisible(map[string]any{"created_at": "2026-06-21T04:00:00Z"}, cols)) - assert.False(t, gt.RowVisible(map[string]any{"created_at": "2026-06-20T04:00:00Z"}, cols)) - - bad := evalRowFilter(t, map[string]Filter{"created_at": {Eq: new("not-a-time")}}, nil) - assert.False(t, bad.RowVisible(map[string]any{"created_at": "2026-06-21T04:00:00Z"}, cols), - "unparseable constant withholds") - - noParser := map[string]ColumnSpec{"created_at": {Kind: ColumnTime}} - assert.False(t, eq.RowVisible(map[string]any{"created_at": "2026-06-21T04:00:00Z"}, noParser), - "ColumnTime without a parser refuses the comparison — never a text fallback") -} - -// TestRowVisible_TimeColumn_OverlongOperand_Gated: the O(1) length gate refuses -// an over-long timestamp operand — string OR json.Number (a Unix-epoch payload -// shape) — BEFORE the parser scans it, so a client-controlled megabyte "value" -// can't stall the per-subscriber fan-out. The parser here counts every call, so -// a gated operand must produce zero calls. -func TestRowVisible_TimeColumn_OverlongOperand_Gated(t *testing.T) { - t.Parallel() - var calls int - counting := ColumnSpec{Kind: ColumnTime, ParseTime: func(v any) (time.Time, bool) { - calls++ - return time.Time{}, false - }} - cols := map[string]ColumnSpec{"created_at": counting} - perms := evalRowFilter(t, map[string]Filter{"created_at": {Eq: new("2026-06-21T04:00:00Z")}}, nil) - - huge := strings.Repeat("9", maxTimeOperandChars+1) - assert.False(t, perms.RowVisible(map[string]any{"created_at": huge}, cols)) - assert.False(t, perms.RowVisible(map[string]any{"created_at": json.Number(huge)}, cols)) - assert.Zero(t, calls, "an over-long operand must be refused before the parser is called") -} - -func TestRowVisible_NilReceiver_AllVisible(t *testing.T) { - t.Parallel() - var perms *ResolvedPermissions - assert.True(t, perms.RowVisible(map[string]any{"x": "y"}, nil)) - assert.False(t, perms.HasRowFilter()) -} - -// TestRowVisible_NonScalar_FailsClosed: a nested/array/null event value can't be -// compared to a scalar filter value, so the predicate fails closed rather than guess. -func TestRowVisible_NonScalar_FailsClosed(t *testing.T) { - t.Parallel() - perms := evalRowFilter(t, map[string]Filter{"tenant_id": {Eq: new("acme")}}, nil) - assert.False(t, perms.RowVisible(map[string]any{"tenant_id": []any{"acme"}}, nil)) - assert.False(t, perms.RowVisible(map[string]any{"tenant_id": nil}, nil)) -} - -// TestRowVisible_MultiplePredicates_AllMustPass: predicates are ANDed, matching the -// query path's "AND"-joined WHERE clause. -func TestRowVisible_MultiplePredicates_AllMustPass(t *testing.T) { - t.Parallel() - perms := evalRowFilter(t, map[string]Filter{ - "tenant_id": {Eq: new("{{ jwt.tenant }}")}, - "amount": {Gt: new("100")}, - }, map[string]any{"tenant": "acme"}) - num := map[string]ColumnSpec{"amount": intNumeric()} - assert.True(t, perms.RowVisible(map[string]any{"tenant_id": "acme", "amount": float64(250)}, num)) - assert.False(t, perms.RowVisible(map[string]any{"tenant_id": "acme", "amount": float64(9)}, num), "amount fails") - assert.False(t, perms.RowVisible(map[string]any{"tenant_id": "globex", "amount": float64(250)}, num), "tenant fails") -} - -// TestRowFilter_UnresolvableClaim_NoRowsOnBothPaths pins the #457 fail-closed -// rule on BOTH read surfaces at once: a filter template whose claim the token -// doesn't carry renders the constant-false predicate on the query path AND -// withholds every row in the stream's in-memory evaluation. One Evaluate -// resolution drives both, so a claim-less token can never see zero rows on -// /v1/query yet every row on /v1/stream. HasRowFilter must stay true for the -// failed predicate — dropping it would put the role back on the unfiltered -// once-per-role fast path, the exact fail-open this test exists to prevent. -func TestRowFilter_UnresolvableClaim_NoRowsOnBothPaths(t *testing.T) { - t.Parallel() - noTenant := map[string]any{"role": "user"} // validly signed token, no tenant claim - tests := []struct { - name string - filter map[string]Filter - claims map[string]any - }{ - {"_eq", map[string]Filter{"tenant_id": {Eq: new("{{ jwt.tenant }}")}}, noTenant}, - {"_neq, the leak direction", map[string]Filter{"tenant_id": {Neq: new("{{ jwt.tenant }}")}}, noTenant}, - {"_gt", map[string]Filter{"tenant_id": {Gt: new("{{ jwt.tenant }}")}}, noTenant}, - {"_in with surrounding text", map[string]Filter{"tenant_id": {In: new("t-{{ jwt.tenant }}")}}, noTenant}, - { - "object claim in a scalar slot", - map[string]Filter{"tenant_id": {Eq: new("{{ jwt.meta }}")}}, - map[string]any{"meta": map[string]any{"tenant": "acme"}}, - }, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - perms := evalRowFilter(t, tt.filter, tt.claims) - assert.Equal(t, "1 = 0", perms.Select.WhereClause, "query path: constant-false predicate") - assert.Empty(t, perms.Select.WhereParams) - assert.True(t, perms.HasRowFilter(), "failed predicate must keep the stream on the per-subscriber path") - assert.False(t, perms.RowVisible(map[string]any{"tenant_id": "acme"}, nil), "stream path: every row withheld") - }) - } -} - -// TestRowVisible_FloatNarrowing: Float32/Float64 columns compare in the -// column's float domain — BOTH operands narrowed, exactly as ClickHouse -// narrows the stored value at insert and the bound constant at compare. The -// Float32 case is the #381 review repro: payload 16777217 stores as 16777216, -// so `_gt: "16777216"` must NOT admit the event (the SQL predicate over the -// stored row is false), while equality against the same spelling matches on -// both surfaces because the constant narrows too. The integer family, by -// contrast, keeps such neighbors distinct. -func TestRowVisible_FloatNarrowing(t *testing.T) { - t.Parallel() - f32 := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericFloat, Bits: 32}}} - f64 := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericFloat, Bits: 64}}} - intCol := map[string]ColumnSpec{"v": intNumeric()} - - gt := evalRowFilter(t, map[string]Filter{"v": {Gt: new("16777216")}}, nil) - assert.False(t, gt.RowVisible(map[string]any{"v": json.Number("16777217")}, f32), - "stored Float32(16777217) is 16777216, not > 16777216 — the ordering fail-open, closed") - assert.True(t, gt.RowVisible(map[string]any{"v": json.Number("16777218")}, f32), - "16777218 is Float32-representable and greater on both surfaces") - assert.True(t, gt.RowVisible(map[string]any{"v": json.Number("16777217")}, intCol), - "an integer column stores 16777217 exactly, so the same event IS greater there") - - eq := evalRowFilter(t, map[string]Filter{"v": {Eq: new("16777217")}}, nil) - assert.True(t, eq.RowVisible(map[string]any{"v": json.Number("16777217")}, f32), - "the constant narrows like the stored value — ClickHouse matches `= '16777217'` too") - assert.True(t, eq.RowVisible(map[string]any{"v": json.Number("16777216")}, f32), - "the Float32-equal neighbor matches on both surfaces — the column type gave that distinction away") - assert.False(t, eq.RowVisible(map[string]any{"v": json.Number("16777216")}, intCol), - "integer storage keeps the neighbors distinct") - - eq64 := evalRowFilter(t, map[string]Filter{"v": {Eq: new("9007199254740993")}}, nil) - assert.True(t, eq64.RowVisible(map[string]any{"v": json.Number("9007199254740992")}, f64), - "Float64 column: 2^53 neighbors collapse in the storage domain, matching SQL") - assert.False(t, eq64.RowVisible(map[string]any{"v": json.Number("9007199254740992")}, intCol)) - - // A magnitude the float domain can't hold (ClickHouse would store ±Inf) - // refuses the comparison — withheld, never a guessed verdict. - overflow := evalRowFilter(t, map[string]Filter{"v": {Gt: new("0")}}, nil) - assert.False(t, overflow.RowVisible(map[string]any{"v": json.Number("1e39")}, f32), - "beyond Float32 range ⇒ withhold") -} - -// TestRowVisible_DecimalScaleTruncation: Decimal columns compare after -// truncating BOTH operands to the column's scale — ClickHouse's cast semantics -// (1.005, 1.006 and 1.009 all store as 1.00 in a Decimal(10,2), and a bound -// constant '1.005' truncates the same way, so `= '1.005'` matches a stored -// 1.00 while `> '1.004'` matches nothing; verified on 25.5 and 26.6). -func TestRowVisible_DecimalScaleTruncation(t *testing.T) { - t.Parallel() - dec2 := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericDecimal, Precision: 10, Scale: 2}}} - - gt := evalRowFilter(t, map[string]Filter{"v": {Gt: new("1.004")}}, nil) - assert.False(t, gt.RowVisible(map[string]any{"v": json.Number("1.005")}, dec2), - "stored 1.00 vs constant 1.00: not greater — the pre-narrowing payload must not leak through") - assert.True(t, gt.RowVisible(map[string]any{"v": json.Number("1.02")}, dec2)) - - eq := evalRowFilter(t, map[string]Filter{"v": {Eq: new("1.005")}}, nil) - assert.True(t, eq.RowVisible(map[string]any{"v": json.Number("1.006")}, dec2), - "both operands truncate to 1.00 — ClickHouse matches this pair too") - assert.False(t, eq.RowVisible(map[string]any{"v": json.Number("1.02")}, dec2)) - - lt := evalRowFilter(t, map[string]Filter{"v": {Lt: new("-1")}}, nil) - assert.False(t, lt.RowVisible(map[string]any{"v": json.Number("-1.005")}, dec2), - "truncation is toward zero: -1.005 stores as -1.00, which is not < -1") -} - -// TestRowVisible_IntegerFractionalOperand_FailsClosed: an integer column -// refuses fractional operands on either side — a fractional constant is a -// per-query type error on the SQL path (the role reads no rows there), and a -// fractional payload was never storable in the column — so the stream -// withholds rather than inventing a verdict SQL cannot produce. An integral -// value in fractional SPELLING is a different thing entirely: it -// canonicalizes to its integer and compares normally. -func TestRowVisible_IntegerFractionalOperand_FailsClosed(t *testing.T) { - t.Parallel() - num := map[string]ColumnSpec{"v": intNumeric()} - - frac := evalRowFilter(t, map[string]Filter{"v": {Neq: new("1.5")}}, nil) - assert.False(t, frac.RowVisible(map[string]any{"v": json.Number("2")}, num), - "fractional constant: SQL errors the query, the stream withholds — neither returns rows") - - pay := evalRowFilter(t, map[string]Filter{"v": {Gt: new("1")}}, nil) - assert.False(t, pay.RowVisible(map[string]any{"v": json.Number("2.5")}, num), - "fractional payload was never storable in an integer column") - assert.True(t, pay.RowVisible(map[string]any{"v": json.Number("2.0")}, num), - "integral value in fractional spelling canonicalizes to 2 and compares") -} - -// TestRowVisible_ConstantSpellingFidelity: on an integer column a literal -// constant that is not its own canonical form ("1e3", "1.5") refuses the -// comparison — the SQL path binds the spelling as written and ClickHouse's -// integer cast errors the query there, so the role reads no rows; admitting -// the canonical reading here would deliver rows SQL never returns. Float and -// Decimal casts accept every JSON-number spelling ('1e3' casts to -// Decimal(10,2) as 1000 — verified against ClickHouse), so those families -// compare such constants normally. -func TestRowVisible_ConstantSpellingFidelity(t *testing.T) { - t.Parallel() - intCol := map[string]ColumnSpec{"v": intNumeric()} - dec2 := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericDecimal, Precision: 10, Scale: 2}}} - f32 := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericFloat, Bits: 32}}} - - exp := evalRowFilter(t, map[string]Filter{"v": {Eq: new("1e3")}}, nil) - assert.False(t, exp.RowVisible(map[string]any{"v": json.Number("1000")}, intCol), - "integer column: SQL errors on '1e3', so the stream must not admit its canonical reading") - assert.True(t, exp.RowVisible(map[string]any{"v": json.Number("1000")}, f32), - "float column: ClickHouse casts '1e3' fine, so the stream compares it") - assert.True(t, exp.RowVisible(map[string]any{"v": json.Number("1000")}, dec2), - "Decimal column: ClickHouse casts '1e3' fine too, so the stream compares it") - - trailing := evalRowFilter(t, map[string]Filter{"v": {Gt: new("1.50")}}, nil) - assert.True(t, trailing.RowVisible(map[string]any{"v": json.Number("2")}, dec2), - "a trailing-zero Decimal spelling casts on both surfaces and compares by value") -} - -// TestRowVisible_OutOfRangeOperand_FailsClosed: an operand outside the -// column's numeric range refuses the comparison on either side. ClickHouse's -// reading of such a CONSTANT was measured to vary by pair on one release — -// error (negative vs unsigned: the role reads no rows), mathematical promotion -// ('256' vs UInt8), or a width-boundary wrap that compares against a DIFFERENT -// value than written (2^63 vs Int64 reads as −2^63, where exact-precision -// comparison would admit the −2^63 rows SQL hides under !=) — so no single -// model is safe to reproduce, and refusal is. A PAYLOAD out of range was never -// storable (the insert is rejected), so withholding matches the stored world. -func TestRowVisible_OutOfRangeOperand_FailsClosed(t *testing.T) { - t.Parallel() - u64 := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericInteger, Bits: 64, Unsigned: true}}} - u8 := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericInteger, Bits: 8, Unsigned: true}}} - i8 := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericInteger, Bits: 8}}} - dec := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericDecimal, Precision: 10, Scale: 2}}} - - neg := evalRowFilter(t, map[string]Filter{"v": {Gt: new("-5")}}, nil) - assert.False(t, neg.RowVisible(map[string]any{"v": json.Number("7")}, u64), - "negative constant on an unsigned column: SQL errors, the stream must not admit everything") - - wrap := evalRowFilter(t, map[string]Filter{"v": {Neq: new("18446744073709551616")}}, nil) - assert.False(t, wrap.RowVisible(map[string]any{"v": json.Number("7")}, u64), - "a width-boundary constant may wrap on the SQL side — comparing it as written risks admitting rows SQL hides") - - wide := evalRowFilter(t, map[string]Filter{"v": {Lt: new("99999999999999999999999")}}, nil) - assert.False(t, wide.RowVisible(map[string]any{"v": json.Number("7")}, u64), - "a constant past the width has no reliable SQL reading; the stream withholds") - - over8 := evalRowFilter(t, map[string]Filter{"v": {Neq: new("256")}}, nil) - assert.False(t, over8.RowVisible(map[string]any{"v": json.Number("7")}, u8)) - assert.False(t, over8.RowVisible(map[string]any{"v": json.Number("300")}, u8), - "an out-of-range payload was never storable — withheld") - ok8 := evalRowFilter(t, map[string]Filter{"v": {Neq: new("254")}}, nil) - assert.True(t, ok8.RowVisible(map[string]any{"v": json.Number("7")}, u8), "in-range operands still compare") - - i8lo := evalRowFilter(t, map[string]Filter{"v": {Gt: new("-129")}}, nil) - assert.False(t, i8lo.RowVisible(map[string]any{"v": json.Number("0")}, i8)) - i8ok := evalRowFilter(t, map[string]Filter{"v": {Gt: new("-128")}}, nil) - assert.True(t, i8ok.RowVisible(map[string]any{"v": json.Number("0")}, i8), "the signed minimum itself is in range") - - prec := evalRowFilter(t, map[string]Filter{"v": {Eq: new("999999999")}}, nil) - assert.False(t, prec.RowVisible(map[string]any{"v": json.Number("5")}, dec), - "9 integer digits exceed Decimal(10,2)'s 8-digit budget: unmodelable on the SQL side, withheld here") - precOK := evalRowFilter(t, map[string]Filter{"v": {Eq: new("99999999")}}, nil) - assert.True(t, precOK.RowVisible(map[string]any{"v": json.Number("99999999")}, dec), "the budget's edge is in range") - - // A hand-built spec with an incoherent Scale must degrade to refusal — - // never a panic on the fan-out goroutine (truncateScale slices by Scale). - badScale := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericDecimal, Precision: 10, Scale: -1}}} - assert.NotPanics(t, func() { - assert.False(t, precOK.RowVisible(map[string]any{"v": json.Number("1.5")}, badScale)) - }) -} - -// TestRowVisible_NumericWithoutStorageModel_FailsClosed pins the NumericSpec -// zero value: a ColumnNumeric spec carrying no storage model must refuse every -// comparison, so a numeric type the classifier doesn't recognize (or a caller -// that forgot to set the model) can never compare under guessed semantics. -func TestRowVisible_NumericWithoutStorageModel_FailsClosed(t *testing.T) { - t.Parallel() - perms := evalRowFilter(t, map[string]Filter{"v": {Eq: new("1")}}, nil) - - unmodeled := map[string]ColumnSpec{"v": {Kind: ColumnNumeric}} - assert.False(t, perms.RowVisible(map[string]any{"v": json.Number("1")}, unmodeled)) - - // A float family with no (or a bogus) bit width must refuse too: - // ParseFloat would silently treat any other bitSize as 64 and compare a - // Float32 column in the wider domain — the fail-open direction. - widthless := map[string]ColumnSpec{"v": {Kind: ColumnNumeric, Numeric: NumericSpec{Family: NumericFloat}}} - assert.False(t, perms.RowVisible(map[string]any{"v": json.Number("1")}, widthless)) -} diff --git a/internal/stream/hub.go b/internal/stream/hub.go index 6683e565..3915c32e 100644 --- a/internal/stream/hub.go +++ b/internal/stream/hub.go @@ -314,21 +314,6 @@ func (h *Hub) rowAdmitted(p *policy.Policy, role string, ev *eventView, claims m return true } -// NumericSpecOf renders discovery's storage classification as the Go row-filter -// evaluator's storage model. The Hub no longer uses it — row filtering is the -// type layer's — and it is kept only for the tests/integration oracle that -// still calls it, until that test and the Go evaluator are removed together. -func NumericSpecOf(st discovery.NumericStorage) policy.NumericSpec { - switch { - case st.Integer: - return policy.NumericSpec{Family: policy.NumericInteger, Bits: st.IntBits, Unsigned: st.Unsigned} - case st.FloatBits != 0: - return policy.NumericSpec{Family: policy.NumericFloat, Bits: st.FloatBits} - default: - return policy.NumericSpec{Family: policy.NumericDecimal, Precision: st.Precision, Scale: st.Scale} - } -} - // eventView is one published event decoded once per Broadcast: cells, the raw // JSON value at each envelope column position, sliced positionally into the // outgoing frame so a value's bytes are never re-encoded. The row-filter reads diff --git a/internal/stream/hub_test.go b/internal/stream/hub_test.go index f6d3b715..46a70dc0 100644 --- a/internal/stream/hub_test.go +++ b/internal/stream/hub_test.go @@ -490,7 +490,7 @@ func col(name, chType string) discovery.Column { } // chtypesHub is a Hub whose row filtering is decided the way production decides -// it: by the type layer, against the real ClickHouse 26.6 artifact, with the +// it: by the type layer, against the real ClickHouse artifact, with the // tables bound for the default tenant. Every row-filter test goes through this // rather than a stub, because the verdicts under test ARE ClickHouse's — // storage-domain narrowing, instant equality across spellings, exactness past From 6a725366510b9b2a164f6a45efb5f05031db44b5 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:36:52 -0400 Subject: [PATCH 20/70] docs: reconcile docs with the chtypes type layer as landed ClickHouse 26.8 and SDK go/v0.5.1 everywhere a current version is named, the generic 503 body, dedupe id absence, RoleTable and IngestWith, 403 for an empty array under a mismatched grant, stream withheld-row reasons and the decline case, a ClickHouse user access section, and one coherent Unreleased CHANGELOG set. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- .github/workflows/README.md | 4 ++-- AGENTS.md | 8 ++++---- CHANGELOG.md | 10 ++++++---- README.md | 2 +- docs/src/content/docs/access-control.mdx | 2 +- docs/src/content/docs/api.md | 16 +++++++++------- docs/src/content/docs/architecture.md | 18 +++++++++--------- docs/src/content/docs/configuration.mdx | 10 ++++++++++ docs/src/content/docs/deployment.md | 14 ++++++++------ docs/src/content/docs/development.md | 2 +- docs/src/content/docs/ingest-pipeline.md | 2 +- docs/src/content/docs/sdk/reference.md | 2 +- 12 files changed, 53 insertions(+), 37 deletions(-) diff --git a/.github/workflows/README.md b/.github/workflows/README.md index 2e9633d2..7b88047d 100644 --- a/.github/workflows/README.md +++ b/.github/workflows/README.md @@ -80,7 +80,7 @@ Queue settings live in the `main branch protection` ruleset's `merge_queue` rule | pnpm store | `pnpm--` | any node job on miss | Store path resolved from pnpm at runtime. docs-build prunes before its save on a key rotation. | | Playwright Chromium | `playwright--` | docs-build | rehype-mermaid renders via headless Chrome at docs build. | | Astro content collections | `astro--` | lint / docs-build | Warm `astro check`/`build` skip unchanged content. | -| chtypes artifact | `chtypes-abi6---` | unit / integration / e2e via `setup-env` (shared) | `~/.cache/chtypes/artifacts/abi6` — the pinned ClickHouse-version `.so` the SDK dlopens; the SDK's default cache is per ABI revision (`abi/-`), and the revision is in both the path and the key prefix so an older SDK's cache is never restored. Measured for the 26.6 line (linux-amd64), pinned before 26.8: **316 MiB on disk, ~0.06 GB stored**, and 85 MiB over the wire on a miss. Fetched by [`scripts/fetch-chtypes.sh`](../../scripts/fetch-chtypes.sh) (`--frozen`, refuses anything `chtypes.lock` doesn't name), which the CLI makes idempotent even on a cache hit — a manifest check, not a re-download. | +| chtypes artifact | `chtypes-abi6---` | unit / integration / e2e via `setup-env` (shared) | `~/.cache/chtypes/artifacts/abi6` — the pinned ClickHouse-version `.so` the SDK dlopens; the SDK's default cache is per ABI revision (`abi/-`), and the revision is in both the path and the key prefix so an older SDK's cache is never restored. Measured on the 26.6 line (linux-amd64) before the pin moved to 26.8: **316 MiB on disk, ~0.06 GB stored**, and 85 MiB over the wire on a miss. Fetched by [`scripts/fetch-chtypes.sh`](../../scripts/fetch-chtypes.sh) (`--frozen`, refuses anything `chtypes.lock` doesn't name), which the CLI makes idempotent even on a cache hit — a manifest check, not a re-download. | | Go build objects (release) | `gobuild-v3---go-release-` | publish-dev (hand-rolled, not `setup-env`) | `~/.cache/go-build` from each `publish-dev` leg's own native `goreleaser build --single-target` (~0.5 GB per arch). `runner.arch` is load-bearing in the key: both legs (`ubuntu-latest`, `ubuntu-24.04-arm`) report `runner.os == 'Linux'`, so without it they'd overwrite the same entry every run — a guaranteed miss on one of them, forever. Same family and key inputs as the CI flavors; `-release` suffix because these objects share nothing with the native test-build flavors. Budget is now **2 arches × 2 generations** of a native single-target cache, not the old 1 × 2 generations of a single 3-target cross-compile cache — still narrower than the pre-chtypes 8-target matrix (Windows/FreeBSD/darwin-amd64 dropped — chtypes publishes no artifact for any of them). | | Go modules (release read) | `gomod-v1--` | nobody — **restore-only** | `publish-dev` reads `ci.yml`'s shared entry from `main`'s scope via `actions/cache/restore`, so its build isn't slowed by a cold module tree. No post-step save, so 0 GB of budget and no risk of a partial write to the shared key. | | CodeQL DB + deps | `codeql-dependencies-*`, `codeql-overlay-base-database-*` | GHAS default setup | **Not ours** — minted by GitHub's default CodeQL setup, not by any workflow in this repo, and not configurable here. ~0.4 GB. Listed so the budget arithmetic below is honest. | @@ -126,7 +126,7 @@ Never add a per-job copy of content that is a pure function of a lockfile — ke `unit`, `integration` and `e2e` link the chtypes SDK (cgo dlopen of a per-ClickHouse-version `.so`/`.dylib`) and need the artifact for the line the test suite dials — today ClickHouse 26.8, matching `tests/integration/setup_test.go`'s pinned container. `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (job-level `env:` on all three) makes `typelayer.TestEngine` `t.Fatal()` if the artifact is missing instead of `t.Skip()`ing — CI must never quietly skip chtypes-backed tests. -`chtypes.lock` (repo root) pins the exact file + sha256 per platform/line; `scripts/fetch-chtypes.sh` wraps the SDK's own CLI with `--frozen --lock chtypes.lock`, so a fetch here can only install what the lock names, never the rolling `artifacts` release. `setup-env`'s `chtypes: "true"` input (see [Cache inventory](#cache-inventory)) restores `~/.cache/chtypes/artifacts/abi6` and always re-runs the fetch script afterward — cheap on a hit (a manifest check, not a re-download) and what turns a restored-but-unverified cache entry back into a hash-checked one every run. A lock is specific to the SDK's ABI revision: a build from another revision is never selected, so after an SDK bump that changes the revision (6 at v0.4.0) `--frozen` fails with `CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED` until the lock is regenerated the same way, and the `abi6` path and key prefix in `setup-env` move with it. Widening the pinned line set is a two-step: `scripts/fetch-chtypes.sh ` locally to update `chtypes.lock`, then add the line to `LOCK_LINES` in that script. Two hash mismatches are possible and they behave differently — don't read one as the other. **Upstream republished a pinned line under a new sha256**: `--frozen` refuses the artifact the lock does not name, and both `Dockerfile.goreleaser`'s fetch and CI's cache-miss fetch fail, until `chtypes.lock` is regenerated per platform (`go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch 26.8 --lock chtypes.lock --platform `, once each for `darwin-arm64`, `linux-amd64`, `linux-arm64`, without `--frozen`) — and `goreleaser-validate.yml`'s image job, which exercises this same fetch on every PR touching `chtypes.lock` or the release workflows, is what surfaces a republish at PR time rather than mid-release. **The cache holds a library the current lock no longer names** (a re-lock landed, so the exact key missed and `restore-keys` handed back the previous generation): this does *not* fail — measured, the CLI reports `is present but hashes … (want …) — replacing` and re-downloads, then the post-job save mints the new generation. So a re-lock costs one cold fetch per Go job on the first run and nothing after. +`chtypes.lock` (repo root) pins the exact file + sha256 per platform/line; `scripts/fetch-chtypes.sh` wraps the SDK's own CLI with `--frozen --lock chtypes.lock`, so a fetch here can only install what the lock names, never the rolling `artifacts` release. `setup-env`'s `chtypes: "true"` input (see [Cache inventory](#cache-inventory)) restores `~/.cache/chtypes/artifacts/abi6` and always re-runs the fetch script afterward — cheap on a hit (a manifest check, not a re-download) and what turns a restored-but-unverified cache entry back into a hash-checked one every run. A lock is specific to the SDK's ABI revision: a build from another revision is never selected, so after an SDK bump that changes the revision (6 at v0.5.1) `--frozen` fails with `CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED` until the lock is regenerated the same way, and the `abi6` path and key prefix in `setup-env` move with it. Widening the pinned line set is a two-step: `scripts/fetch-chtypes.sh ` locally to update `chtypes.lock`, then add the line to `LOCK_LINES` in that script. Two hash mismatches are possible and they behave differently — don't read one as the other. **Upstream republished a pinned line under a new sha256**: `--frozen` refuses the artifact the lock does not name, and both `Dockerfile.goreleaser`'s fetch and CI's cache-miss fetch fail, until `chtypes.lock` is regenerated per platform (`go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch 26.8 --lock chtypes.lock --platform `, once each for `darwin-arm64`, `linux-amd64`, `linux-arm64`, without `--frozen`) — and `goreleaser-validate.yml`'s image job, which exercises this same fetch on every PR touching `chtypes.lock` or the release workflows, is what surfaces a republish at PR time rather than mid-release. **The cache holds a library the current lock no longer names** (a re-lock landed, so the exact key missed and `restore-keys` handed back the previous generation): this does *not* fail — measured, the CLI reports `is present but hashes … (want …) — replacing` and re-downloads, then the post-job save mints the new generation. So a re-lock costs one cold fetch per Go job on the first run and nothing after. ## Timing (steady state, full pipeline) diff --git a/AGENTS.md b/AGENTS.md index c95c3c82..badd6366 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -38,7 +38,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`coord/`** — leases for work that must run in one process at a time (`Observer.Held` reads whether one is held without campaigning): `Coordinator.TryAcquire(ctx, name)` → a `Term` (fencing `Token`, strictly increasing per name; `Done`/`Err`, `ErrLost` on loss; `Resign`), `ErrHeld` while another holder's — or this coordinator's own — term is live; `RunElected` runs a loop only while holding its lease, resigning when the loop returns and campaigning again every `RetryPeriod`. `Local` is the in-process implementation (first taker wins, never expires; `Peer` is a second handle over the same table for tests); every implementation runs `coordtest.Conformance`. Imports only the standard library, so a distributed backend lives beside its connection: `coord.backend: nats` is `internal/mq/lease.go` (`ExternalNATS.Leases`), a key per lease in the operator's KV bucket, the KV revision as the fencing token, and expiry judged on the candidate's own clock (the same revision seen unchanged for 15s), never by a server TTL. `internal/app`'s `wireCoord` opens the one `coord.backend` selects and the sweeper runs through `RunElected` under the `sweeper` lease - **`dedupe/`** — `Deduplicator` interface (two-phase `Reserve`/`Commit`/`Release` over `Key{Table, ID}`; every backend passes the `dedupetest` conformance suite) → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant and table, pending claims in memory, committed ids stored with their expiry and deleted by an hourly background sweep along with the version-0 keys from before the table joined the key, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges) or `Dynamo` (one shared DynamoDB table, conditional `PutItem` claims; conformance-tested against dynamodb-local, selected by `dedupe.backend: dynamodb`; boot checks the table and never creates it outside dynamodb-local), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` or, gated on the table check (`Factory.Gated`), `Dynamo.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) - **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`, and starts a tenant over on a fresh registry when a reload moves it to another address or database: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version and default timezone, and fires an `OnRefresh` hook after every successful refresh and before the registry reports itself loaded, so a loaded tenant is a bound one — `typelayer.Engine.Bind` is its only consumer (Key Design Decision #21) -- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else; `row` is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced for that stored record (via `typelayer.Table.Ingest`), `columns` names its positions (the table's insertable columns, or the narrower list a column-restricted role produces); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs with `typelayer.InsertSettings()` plus `async_insert=0` (it never loads the artifact, so an ingest-only process needs none). In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) +- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else; `row` is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced for that stored record (via `typelayer`'s `Table.IngestWith`), `columns` names its positions (the table's insertable columns, or the narrower list a column-restricted role produces); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs with `typelayer.InsertSettings()` plus `async_insert=0` (it never loads the artifact, so an ingest-only process needs none). In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) - **`keyenc/`** — the one escaping composite keys are built from: `Escape` keeps `[A-Za-z0-9_-]` (exactly the tenant-id grammar, so a tenant id is its own escaped form) and writes every other byte as `%XX`, `Unescape` is `url.PathUnescape` (lenient: either hex case, and a byte left unescaped reads as itself, so a `%2D` an earlier build wrote still reads), `Join`/`AppendJoin` escape each field and put a separator between them (they panic on no fields, and on a separator the escaping could write or one outside ASCII) and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to and no field reaches a key unescaped: NATS subjects (`Join`/`Split` after the verbatim tenant) and the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — and the dedupe keys (`/
/`) use it; changing what it keeps orphans every stored key (an orphaned dedupe key lets a seen id through again), and on the shared backend, whose keys every process builds for itself, splits them between builds for the length of a rolling upgrade (a bump one build makes misses the entries the other filed, served until their TTL) - **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20), and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal, `ErrUnavailable` a broker that cannot be reached — both a `503`, with `Retry-After` `30` and `5`; `WithIdempotencyKey` makes a republish inside the queue's duplicate window a no-op), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). A broker whose ingest queue is split into units one consumer at a time owns implements `Sharded` too (`IngestUnits`, `ResetOrphaned`, `Unowned`; `ConsumerConfig.Units` narrows a consumer to some of them, and its consumer is a `Releaser` and a `Halter`, capped per unit by `ConsumerConfig.MaxHeld`, with `ErrConsumerMismatch` when an operator durable no longer fits), and `Message.OnSettled` runs a hook once, at the first ack or nak attempt, confirmed or not. Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, stream names, sequences, and ack floors are private to the implementations, whose subject tokens are escaped by the shared `internal/keyenc`: `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`, `deadletter.go`), which `internal/app` constructs and hands everything else as a `mq.Broker`, and `ExternalNATS` (`external.go`, `subject_nats.go`, `nats_topology.go`: an operator-owned cluster whose streams, durables and lease bucket it never creates, changes, purges or deletes; `lease.go` holds `coord.backend: nats`'s leases in that bucket), which `internal/app` constructs from the `mq.nats` block when `mq.backend` is `nats` ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every implementation passes the conformance suite in `internal/mq/mqtest` (`mqtest.Run`), which states the `Broker` contract as behavior; a new backend runs it from its own test, with `mqtest.Caps` only where its semantics legitimately differ - **`observability/`** — OpenTelemetry pipeline: `InitProvider` wires trace/metric/log providers via OTLP gRPC (each signal independently gated). A top-level `Prometheus` config block drives an optional `/metrics` scrape endpoint that runs independently of OTLP push — standalone (Alloy/Mimir scrape, no collector), alongside OTLP, or off. `NewLogger` produces a slog handler that fans out to stdout AND OTLP (stdout always 100%, OTLP sample-rate-aware). `TraceHandler` injects trace_id/span_id from active spans. `tracer.go` provides W3C trace context propagation over message headers (`InjectHeaders`/`ExtractHeaders` on a plain header map; `internal/mq` injects on every publish and extracts on the `Subscribe` path, so this package never sees a NATS type). @@ -47,7 +47,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes - **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row via `typelayer`, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes` (the sole exception: `cmd/wavehouse/main.go` references `typelayer` itself). One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns, plus a `DEFAULT ''` per `_eq` check column — which is how column policy and auto-inject are answered with no Go-side record inspection. `Table.Ingest(format, body)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) +- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes` (the sole exception: `cmd/wavehouse/main.go` references `typelayer` itself). One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns, plus a `DEFAULT ''` per `_eq` check column — which is how column policy and auto-inject are answered with no Go-side record inspection. `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) - **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -72,9 +72,9 @@ The invariant index — what must stay true. Full narrative and rationale live i 16. **Bearer-token-only CORS posture (security)** — Bearer JWT on every request, no cookies/sessions; `corsMiddleware` deliberately **never** emits `Access-Control-Allow-Credentials` (not needed, and `*` + credentials is a spec violation browsers reject). `cors.allowed_origins` (settings directory, per tenant: a tenant route is decorated from the list of the tenant it names, everything else from tenant `0`'s — `corsOrigins`) controls who can *read* responses, not cookie scope; CSRF protection is structural. Don't reintroduce cookie auth or `Allow-Credentials` without a design discussion — answers GitHub #29/#30. Code: `internal/api/router.go`. 17. **Non-fatal boot** — schema-discovery failure on boot is non-fatal: `internal/app` records an `api.BootState`, binds `:8080`, serves 503 on `/livez`/`/readyz` with the diagnostic, and retries via `SchemaRegistry.RetryRefresh` (jittered backoff 2s → 60s), per tenant over a nested directory: `/livez` is 503 while no tenant has completed a first discovery, then sticky 200, and a tenant's outage after that is its log line and counter, never a probe failure. Until a tenant's first discovery its table lookups are a 503 with `Retry-After`, not a 404. Bounds supervisor restart loops. 18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention; `/readyz` pings every open ClickHouse pool at once and is ready at the first answer, 503 naming each when none answers); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. -19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer.Table.Ingest`, via chtypes), in the column's declared zone else the server's default (`"2026-06-21 04:00:00.123"`, never RFC 3339's `Z` suffix), and `/v1/query` and pipes are rendered by the same ClickHouse in the same spelling — there is no separate WaveHouse rewrite step to keep in sync, so live and query reads can't drift on spelling *or* instant (#372) while the process's chtypes zone equals the zone the server renders in. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. +19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), in the column's declared zone else the server's default (`"2026-06-21 04:00:00.123"`, never RFC 3339's `Z` suffix), and `/v1/query` and pipes are rendered by the same ClickHouse in the same spelling — there is no separate WaveHouse rewrite step to keep in sync, so live and query reads can't drift on spelling *or* instant (#372) while the process's chtypes zone equals the zone the server renders in. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. 20. **Sealed MQ boundary** — only `internal/mq` imports NATS/JetStream (`github.com/nats-io/…`), enforced by the `depguard` rule in `.golangci.yml`, so `make lint` fails on a leak in every package it builds (the `integration`-tagged files under `tests/` are outside lint's build context — keep them clean by convention, through `mq.Broker`). A test outside `internal/mq` that needs a real NATS server goes through `internal/mq/natstest`, which stands one up from the shipped `deployments/nats` files and hands back a URL and passwords, never a NATS type. The boundary is semantic as well: everything else addresses events by `mq.Topic` and states intent through mq-owned interfaces (`Publisher`, `Consumer`, `DeadLetterer`, `Purger`, `Replayer`, …), and never builds a subject, names a stream, or reasons in sequences — so a subject, stream, or broker change lands in one package ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 4; story 5's tenant token landed there alone — `Topic.Tenant`, first in every subject). Don't add a raw accessor (`JetStream()`, `NatsConn()`, `GetServer()`) back, and don't hand-build `"ingest."`/`"dlq."` subjects outside `internal/mq` — widen the mq surface with an intent-level method instead. -21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503`, the stream withholds every row with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}`. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. +21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503` (generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds every row with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. ## Code Conventions diff --git a/CHANGELOG.md b/CHANGELOG.md index 8eb238cb..d08698f6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -59,11 +59,11 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Logging goes through the `slog` default logger; no constructor takes a `*slog.Logger` anymore** (`internal/mq/embedded.go`, `internal/api/{ingest,pipes,structured_query,dlq,settings,errors,router}.go`, `internal/auth/auth.go`, `internal/discovery/{discovery,timestamp}.go`, `internal/ingest/{sweeper,worker}.go`, `internal/settings/{store,watch}.go`, `internal/chconn/chconn.go`, `internal/config/persistence.go`, `internal/app/wire.go`, `internal/testutil/logtest/` (new, + tests), `internal/testutil/testutil.go`): the general-notes refactor of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) and the cleanup deferred from [#586](https://github.com/Wave-RF/WaveHouse/pull/586), which left `internal/mq` logging half through an injected logger and half through the default. The logger parameter or field is gone from `mq.NewEmbedded`, `api.NewIngestHandler` / `NewPipesHandler` / `NewStructuredQueryHandler` / `NewDLQHandler` / `NewSettingsHandler`, `api.RequireAdmin` and `api.Dependencies.Logger`, `auth.NewAuthenticator`, `discovery.NewSchemaRegistry`, `ingest.NewSweeper` and the ingest worker, `settings.Open`, `chconn.Open` (whose field was never read), and `config.WarnIfFreshDataDir` / `LogStorageInitError`. Call sites use the context-aware calls (`slog.ErrorContext(ctx, …)`) wherever a context is in scope, so the trace handler can stamp them. Two visible differences: the ingest worker's lines no longer carry `component=ingest_worker`, and the auth middleware's operator-key audit lines and the settings reload lines are no longer skippable by passing a nil logger (only tests did). Tests reach log output through the new `internal/testutil/logtest`: `Silence()` from a package's `TestMain`, and `Capture(t, level)` for a test that asserts on log lines — which therefore runs serially, since the default logger is process-wide. `testutil.NopLogger` is removed. - **The process wiring moves out of `main.go` into `internal/app`** (`internal/app/` (new: `app.go`, `wire.go`, + tests), `cmd/wavehouse/main.go` (+ tests), `tests/integration/setup_test.go`, `.testcoverage.yml`, `.github/labeler.yml`): `app.New` builds every component from the boot config and the settings directory, `Run` drives the long-lived ones — ingest worker, sweeper, hub bridge, keepalive wheel, schema refresh, SIGHUP and the directory watcher, the API server and the Prometheus sidecar — under one `errgroup` until the signal context is cancelled or one of them fails, and `Close` releases what `New` opened in reverse order. Each component is wired in one place — what it opens, what it loops, what it releases — with the settings store handed to its wiring function whole, so the per-tenant registry ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)) lands there rather than in `main`. `main.go` shrinks to argv dispatch, the logger, `config.Load`, `CheckDataDir`, `app.New`, `app.Run`; `run(ctx)` takes the context `main` cancels on the first `SIGINT`/`SIGTERM`, so it is unit-tested end to end and the per-suite coverage exclude for it is gone. The integration suite boots through the same `app.New` against its testcontainer (the seed settings patched to the container, `default_role` set to the admin role) instead of a hand-built handler subset that had drifted from the binary. Behavior is unchanged except for the stop, which is now bounded end to end in three phases whose budgets add rather than multiply — `server.shutdown_timeout` for the drain, then fixed 5s and 3s for the release and the telemetry flush, so the worst case is the timeout plus 8s: `Run` drains the ingest worker and the API server's in-flight requests — the process could previously exit while the worker drain was still in flight — while open SSE streams are ended the moment the drain begins (a stream is a connection to close, not work to wait for; the client reconnects via `Last-Event-ID`) instead of holding the stop for the whole timeout; then `Close(ctx)` releases the stores under a context of its own, so a remote store's close can give up at the deadline rather than hang the exit, and finally flushes telemetry on a separate short budget so the flush that reports on the stop is never starved by a slow close. A settings reload caught mid-hook by the stop gives up with it. A second `SIGTERM`/`SIGINT` during the stop abandons it and exits non-zero, and `SIGHUP` is ignored once a stop has begun (it was briefly fatal: the reload loop's `signal.Stop` restored the default disposition at the start of the drain). The `mq.max_bytes_gb` reload hook bounds its JetStream calls to ten seconds, and when the DLQ resize fails it rolls the ingest stream back under a budget of its own instead of the one that just expired. `deployments/compose/standalone.yaml` sets `stop_grace_period` to cover all three phases, and the deployment docs gain a [Stopping](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md#stopping) section. Closes [#140](https://github.com/Wave-RF/WaveHouse/issues/140); story 0 of #583. -- **The chtypes SDK is `go/v0.4.0` (ABI revision 6)** (`go.mod`, `chtypes.lock`, `scripts/fetch-chtypes.sh`, `.github/actions/setup-env/action.yml`): the lock pins the revision-6 26.6.8.7 build (`b1790767905`) on darwin-arm64, linux-amd64 and linux-arm64, and must be regenerated whenever the SDK's ABI revision changes. The default artifact cache moved to `~/.cache/chtypes/artifacts/abi6/-`, so the first fetch after upgrading downloads again (explicit `--dest` / `CHTYPES_REGISTRY` directories, as the Docker images use, are unaffected); CI's cache key and path carry the revision. Insert `check` clauses now cost one parse instead of two: they are judged inside the same `RowsExportWith` call that validates the body, through a compiled row filter, and the five ingest formats (JSON family, CSV, TSV, CSV and TSV with `header=present`) all take that path. +- **The chtypes SDK is `go/v0.5.1` (ABI revision 6), and the supported ClickHouse line is 26.8** (`go.mod`, `chtypes.lock`, `scripts/fetch-chtypes.sh`, `.github/actions/setup-env/action.yml`, `deployments/compose/*.yaml`, `tests/integration/setup_test.go`): the lock pins the revision-6 26.8.15.10 build (`b1790845279`) on darwin-arm64, linux-amd64 and linux-arm64, and must be regenerated whenever the SDK's ABI revision changes. The compose files, CI and the integration suite run `clickhouse/clickhouse-server:26.8.15.10`, off the retired 26.6 line. One 26.8 rule worth knowing: a bare JSON number in a `DateTime64` column is read as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — ClickHouse's rule, which WaveHouse stores as the server would. The default artifact cache moved to `~/.cache/chtypes/artifacts/abi6/-`, so the first fetch after upgrading downloads again (explicit `--dest` / `CHTYPES_REGISTRY` directories, as the Docker images use, are unaffected); CI's cache key and path carry the revision. Insert `check` clauses now cost one parse instead of two: they are judged inside the same `RowsExportWith` call that validates the body, through a compiled row filter, and the five ingest formats (JSON family, CSV, TSV, CSV and TSV with `header=present`) all take that path. - **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `builds:` and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes --notes-start-tag "$(git describe --match 'v*')"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. -- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.4.0, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/hub.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, beside the string `code` class the other error bodies carry; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` results — carry ClickHouse's own rendering** (`"2026-06-21 04:00:00.123"`, in the column's zone) instead of the RFC 3339 `Z`-suffixed form WaveHouse used to canonicalize to, by construction rather than by a rewriting step (closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5`, and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.1, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` results — carry ClickHouse's own rendering** (`"2026-06-21 04:00:00.123"`, in the column's zone) instead of the RFC 3339 `Z`-suffixed form WaveHouse used to canonicalize to, by construction rather than by a rewriting step (closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. - **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, without the columns it may not write, instead of walking a decoded record's keys. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a checked column behaves exactly like omitting it. @@ -83,9 +83,9 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` / `InsertableColumns` / `InsertableColumnNames`, memoized per table since the ingest path would otherwise rebuild them once per record. `EPHEMERAL` stays insertable — it is insert-only by construction, never stored, confirmed on the same server rather than assumed. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. -- **Ingest reads the request body up front, and the per-record decisions sit behind interfaces** (`internal/api/{ingest,ingest_seams,bufpool,record_reader}.go`, `internal/stream/hub.go`, `internal/ingest/compact.go`): responses are unchanged except at the body cap and one new read-failure body (`400 {"error":"invalid request body"}`, when the body cannot be read at all — a malformed transfer encoding or a truncated upload, which previously surfaced through the decoder as `invalid json`), and at the cap the `413` is now decided before any record is processed: an over-cap batch no longer ingests the prefix it had already decoded, and an over-cap single-object body whose first object was followed by an oversized tail — which used to answer `200` after ingesting that one object — now answers `413`. Both are improvements, since a client retrying a `413` can no longer double-insert a prefix, but they are behavior changes and the memory profile changes too (see below); this is the seam work the native type layer lands against. The handler now reads the whole (already `MaxBytesReader`-capped) body into a pooled `*bytes.Buffer` and runs the record readers over those bytes rather than the live connection — so the `413` surfaces at that read instead of mid-iteration (same status, same message), and the `415` is decided from the header before a single byte is read. Three decision points became interfaces with default implementations that delegate to today's code unchanged: `RecordValidator` (schema validation + timestamp canonicalization — the two calls stay where they are, with the check-clause block between them, since merging them would move checks onto canonicalized values), `InsertChecker` (the `_eq` and `_in` comparisons), and `stream.RowEvaluator` (row visibility, reached by both the live fan-out and replay through the one shared admission step). **The memory profile is not unchanged, and that is the deliberate part.** Streaming meant peak resident bytes on the order of one record: the NDJSON path scanned line by line and the array path let `json.Decoder` compact after each element. Peak is now O(body) per in-flight request — and `bytes.Buffer` grows by doubling, so the peak allocation can exceed the body cap before `MaxBytesReader` errors. `maxPooledBufferBytes` (1 MiB) caps what a request hands *back* to the pool, not its peak, and nothing in `internal/api` bounds total in-flight bytes, so the ceiling is concurrency × the 16 MiB data-plane cap — which has no operator knob (`maxRequestBytes` is test-only), so the outer limit is the reverse proxy's, which the reverse-proxy guide already advises setting. Kept because it is the shape the native type layer lands against, which needs the body addressable rather than consumed; a bound on total in-flight ingest bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). Operators fronting large batches at high concurrency should size for it or cap body size at the proxy. All three are nil-safe: an un-wired handler or `Hub` uses the default rather than panicking past the check. Also new: `ingest.EncodeCompactRow`, which renders a record as one `JSONCompactEachRow` line — inert in this commit, and the encoder every published row goes through by the end of the release. +- **Ingest reads the request body up front, and the per-record decisions sit behind interfaces** (`internal/api/{ingest,ingest_seams,bufpool,record_reader}.go`, `internal/stream/hub.go`, `internal/ingest/compact.go`): responses are unchanged except at the body cap and one new read-failure body (`400 {"error":"invalid request body"}`, when the body cannot be read at all — a malformed transfer encoding or a truncated upload, which previously surfaced through the decoder as `invalid json`), and at the cap the `413` is now decided before any record is processed: an over-cap batch no longer ingests the prefix it had already decoded, and an over-cap single-object body whose first object was followed by an oversized tail — which used to answer `200` after ingesting that one object — now answers `413`. Both are improvements, since a client retrying a `413` can no longer double-insert a prefix, but they are behavior changes and the memory profile changes too (see below); this is the seam work the native type layer lands against. The handler now reads the whole (already `MaxBytesReader`-capped) body into a pooled `*bytes.Buffer` and runs the record readers over those bytes rather than the live connection — so the `413` surfaces at that read instead of mid-iteration (same status, same message), and the `415` is decided from the header before a single byte is read. Three decision points became interfaces with default implementations that delegate to today's code unchanged: `RecordValidator` (schema validation + timestamp canonicalization — the two calls stay where they are, with the check-clause block between them, since merging them would move checks onto canonicalized values), `InsertChecker` (the `_eq` and `_in` comparisons), and `stream.RowEvaluator` (row visibility, reached by both the live fan-out and replay through the one shared admission step). **The memory profile is not unchanged, and that is the deliberate part.** Streaming meant peak resident bytes on the order of one record: the NDJSON path scanned line by line and the array path let `json.Decoder` compact after each element. Peak is now O(body) per in-flight request — and `bytes.Buffer` grows by doubling, so the peak allocation can exceed the body cap before `MaxBytesReader` errors. `maxPooledBufferBytes` (1 MiB) caps what a request hands *back* to the pool, not its peak, and nothing in `internal/api` bounds total in-flight bytes, so the ceiling is concurrency × the 16 MiB data-plane cap — which has no operator knob (`maxRequestBytes` is test-only), so the outer limit is the reverse proxy's, which the reverse-proxy guide already advises setting. Kept because it is the shape the native type layer lands against, which needs the body addressable rather than consumed; a bound on total in-flight ingest bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). Operators fronting large batches at high concurrency should size for it or cap body size at the proxy. All three are nil-safe: an un-wired handler or `Hub` uses the default rather than panicking past the check. Also new: `ingest.EncodeCompactRow`, which renders a record as one `JSONCompactEachRow` line — inert in this commit, and the encoder every published row went through at that point in the release (superseded: the published row is now ClickHouse's own export, and `EncodeCompactRow`, `RecordValidator` and `InsertChecker` are gone — see the type-layer entry). -- **NATS envelope v2: the row travels positionally, with the column names sent alongside** (BREAKING; `internal/ingest/{types,compact,worker}.go`, `internal/api/ingest.go`, `docs/src/content/docs/{api.md,architecture.md,ingest-pipeline.md}`): `EventMessage`'s `data` object is replaced by `format` (`"JSONCompactEachRow"`), `columns` (the table's declaration order) and `row` (one compact line — a positional JSON array). The `INSERT` the worker emits carries the column names once for a whole group instead of every row repeating every key (each NATS envelope still carries its own `columns`, since a message must stand alone), and a reader can tell a schema change mid-stream from a reordering. **In-flight NATS messages published by an older version are not readable by the new worker** — an envelope whose `format` is absent or unknown, or whose `columns` and `row` can't be paired (a length mismatch, an undecodable row), carries no way to say which value belongs to which column. Such an envelope is parked on the DLQ (see the Fixed entry below), never inserted. **Drain the ingest queue before deploying.** The worker groups a batch by column list, so a schema change mid-stream splits the INSERT rather than corrupting it, and writes `INSERT INTO {table} (cols) FORMAT JSONCompactEachRow` — the table still binds as a server-side `Identifier` parameter, while the column list, which has no such parameter, is quoted client-side by the same `chsql.QuoteIdent` the query builder uses. A positional row has one value per column and no way to say "absent", so a field the record omitted now rides as an explicit `null` in its slot; `input_format_null_as_default=1` (already the server default — set explicitly for one configured otherwise) turns that back into the column's default for a **non-nullable** column, matching what omitting the key did under `JSONEachRow`. **Transitional divergence, on `Nullable` columns only, and it is not what that setting controls:** ClickHouse stores an explicit `null` as `NULL` on a nullable column whatever the setting says — only an *absent* key ever took the default — so a `Nullable(T) DEFAULT …` column now stores `NULL` where it previously took its default. Verified against ClickHouse 26.6.3 (omitted key → default; explicit null → `NULL` at either setting). *Against a server explicitly running `input_format_null_as_default=0`*, the reverse also changes: an explicit `null` for a non-nullable column with a default now takes the default rather than failing the row into the DLQ, because WaveHouse pins the setting instead of inheriting it. On a default-configured server that was already the behavior. Every other column type behaves as before. The DLQ flow is unchanged — its payload is the new envelope. Row cells are copied as their original bytes rather than re-encoded at each hop, so a 64-bit id past 2^53 keeps every digit end to end. +- **NATS envelope v2: the row travels positionally, with the column names sent alongside** (BREAKING; `internal/ingest/{types,compact,worker}.go`, `internal/api/ingest.go`, `docs/src/content/docs/{api.md,architecture.md,ingest-pipeline.md}`): `EventMessage`'s `data` object is replaced by `format` (`"JSONCompactEachRow"`), `columns` (the table's declaration order) and `row` (one compact line — a positional JSON array). The `INSERT` the worker emits carries the column names once for a whole group instead of every row repeating every key (each NATS envelope still carries its own `columns`, since a message must stand alone), and a reader can tell a schema change mid-stream from a reordering. **In-flight NATS messages published by an older version are not readable by the new worker** — an envelope whose `format` is absent or unknown, or whose `columns` and `row` can't be paired (a length mismatch, an undecodable row), carries no way to say which value belongs to which column. Such an envelope is parked on the DLQ (see the Fixed entry below), never inserted. **Drain the ingest queue before deploying.** The worker groups a batch by column list, so a schema change mid-stream splits the INSERT rather than corrupting it, and writes `INSERT INTO {table} (cols) FORMAT JSONCompactEachRow` — the table still binds as a server-side `Identifier` parameter, while the column list, which has no such parameter, is quoted client-side by the same `chsql.QuoteIdent` the query builder uses. A positional row has one value per column and no way to say "absent", so a field the record omitted now rides as an explicit `null` in its slot; `input_format_null_as_default=1` (already the server default — set explicitly for one configured otherwise) turns that back into the column's default for a **non-nullable** column, matching what omitting the key did under `JSONEachRow`. **Transitional divergence, on `Nullable` columns only, and it is not what that setting controls:** ClickHouse stores an explicit `null` as `NULL` on a nullable column whatever the setting says — only an *absent* key ever took the default — so a `Nullable(T) DEFAULT …` column now stores `NULL` where it previously took its default. Verified against ClickHouse 26.6.3 (omitted key → default; explicit null → `NULL` at either setting). *Against a server explicitly running `input_format_null_as_default=0`*, the reverse also changes: an explicit `null` for a non-nullable column with a default now takes the default rather than failing the row into the DLQ, because WaveHouse pins the setting instead of inheriting it. On a default-configured server that was already the behavior. Every other column type behaves as before. The DLQ flow is unchanged — its payload is the new envelope. Row cells are copied as their original bytes rather than re-encoded at each hop, so a 64-bit id past 2^53 keeps every digit end to end. (Superseded for the omitted-field behaviour: the row is now ClickHouse's own export with `DEFAULT`s already evaluated, so a `Nullable(T) DEFAULT …` column takes its default again and the divergence above no longer exists — see the type-layer entry.) - **SSE: the column list is announced as an `event: schema` frame, and rows arrive positionally** (BREAKING for raw consumers; `internal/stream/{hub,subscriber,metrics}.go`, `internal/api/stream.go`, `clients/ts/src/stream/sse.ts`, `docs/src/content/docs/{api.md,sdk/streaming.md}`): a data frame's `data` object is replaced by `row`, the compact array reduced to the caller's projected positions, and each connection is sent `event: schema` — `{"table_name", "columns"}` — before its first row and again whenever the projected list changes. The announcement is **per connection**: it goes out at subscribe time from the schema registry, so a client on a quiet table knows the shape before any row arrives, and a late joiner or a reconnect is told again. It deliberately carries **no** `id:` line — an empty one would clear the client's `Last-Event-ID` and cost the connection its resumption point, and a schema frame has no event position of its own to offer. Replay follows the identical contract with its own drift state, because it writes straight to the socket while live events queue behind it — sharing the connection's state would let a live announcement claim the slot and leave a replayed row ahead of it with nothing to zip against. **Known limitation, deferred to the schema-versioning work:** those two states are not reconciled when they disagree, so if a table's column set changes while a client is connected *and* that client gap-fill-replays across the change, live rows arriving after the replay may carry no fresh schema frame until the next drift or a reconnect ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The SDK's arity check catches most of it — a row whose **length** disagrees with the announced list is dropped rather than guessed at — but not a **same-length** change (a `RENAME COLUMN`, or a drop paired with an add), where values zip under the wrong names until the next announcement or a reconnect. So the residual case costs correctness, not only availability. The TypeScript SDK consumes the schema event and zips each row back into an object, so `.stream()`, `.liveQuery()` and `StreamEvent.data` are **unchanged** — two things are visible: a column the producer omitted now arrives as an explicit `null` rather than an absent key, and a row object has a **null prototype** (`Object.create(null)`), so a ClickHouse column legitimately named `__proto__` becomes an own property instead of vanishing into the inherited setter — at the cost of `row.hasOwnProperty(…)`, `` `${row}` `` and `row.constructor` no longer working on it. Use `Object.hasOwn(row, …)`. The client-side `.select(…)` projection now builds its row the same way: `projectColumns` used a plain object literal, so a `__proto__` column survived an unprojected stream and vanished from a projected one — the guarantee held for the transport but not for the path most callers use. **The SSE reader and writer changed together, so a version-skewed pair is a silently dead stream — upgrade both.** `@wavehouse/sdk` publishes independently of the server, so a pinned frontend against a self-scheduled backend is the normal shape, not an edge case. A **new SDK against an older server** never receives an `event: schema` frame, so `_columns` stays unset and every data frame is dropped — no `error` callback fires, the stream simply delivers nothing. The warnings are bounded (three per cause per connection, then a suppression line), so **a quiet console is not evidence the stream is healthy**. An **older SDK against a new server** reads the removed `data` key and yields `data: undefined` for every event. Neither surfaces as a catchable error. A **raw** SSE consumer (a hand-rolled `EventSource`) must keep the announced list and zip against it; the SDK drops — with a warning, never an error — a row it has no list for or whose length disagrees with one, rather than delivering values under guessed names. @@ -99,6 +99,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Removed +- **The Go-side validation, timestamp canonicalization, row-filter evaluation and compact encoding** (`internal/discovery/{validation,timestamp}.go`, `internal/policy/{rowfilter,numeric}.go`, `internal/ingest/compact.go`, `internal/api/ingest_seams.go`, `internal/stream/hub.go`): `discovery.Validate` and `CanonicalizeTimestamps`, `policy.RowVisible`, `ingest.EncodeCompactRow`, the `RecordValidator`/`InsertChecker` seams and `stream.NumericSpecOf` are gone. Their jobs — parsing a record, rendering a timestamp, deciding a row filter, writing the positional row — are ClickHouse's own now, through chtypes (see the type-layer entry in Changed). `typelayer.InsertSettings()` is the one static set of parsing settings the worker's `INSERT` and the ingest parse share. + - **`policy.LiteralValue` and `policy.CanonicalNumericLiteral`** (`internal/policy/{canonical,policy}.go`): the marker type and the numeric re-reading of a policy-authored insert-check literal. Insert checks are chtypes filters now, so a literal binds as written and ClickHouse reads it under the column's type — there is no second, numeric reading at compare time. A `_eq: "1.0"` against a `UInt64` used to admit a stored `1`; it is now ClickHouse's code 53 `TYPE_MISMATCH` per row (`422`), and the fix is to write a literal the column can read. `CanonicalScalar` stays: it is still the one rendering layer for a JWT claim. The released-version entry further down this file describing `LiteralValue` as shipped behaviour is left as history. - **The policy's entire HTTP surface — `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, and the SDK's `wh.policy` namespace** (`internal/api/policy.go` + `policy_test.go` (deleted), `internal/api/{router,router_test}.go`, `cmd/wavehouse/main.go`, `clients/ts/src/policy.ts` (deleted), `clients/ts/src/{client,types,index}.ts`, `tests/e2e/sdk/{admin,query,ingest,streaming}.test.ts` + `settings.ts`, `docs/src/content/docs/{api.md,access-control.mdx,settings-directory.mdx,architecture.md,configuration.mdx,development.md,reverse-proxy.mdx,sdk/admin.md,sdk/reference.md}`, `AGENTS.md`; closes [#514](https://github.com/Wave-RF/WaveHouse/issues/514)): both endpoints were born alongside `PUT /v1/ops/policy` and outlived it when [#508](https://github.com/Wave-RF/WaveHouse/pull/508) deleted the policy write API; with files as the only write path, the policy is read, edited, and validated where it lives, so the whole read/dry-run surface goes too. The dry run had also kept its original lenient decoder while adoption became strict, certifying `{"valid": true}` for documents a reload would refuse — a misspelled operator key (`"eq"` for `"_eq"`) silently dropped into a filter that disables row security, the exact fail-open [#460](https://github.com/Wave-RF/WaveHouse/issues/460) demonstrated; deleting it removes the last non-strict policy decode site, closing #514 (the other five sites it cites were deleted or made strict by #508). Its replacement is `wavehouse validate`, which enforces strictly more (the cross-file role references against `roles.json` were invisible to a single-document dry run). The break-glass story narrows accordingly: the operator key can still trigger `POST /v1/ops/settings/reload`, whose findings report exactly why a rejected directory was refused — what it can no longer do is read back the adopted snapshot over HTTP; a bad edit still never breaks a running server (the previous good snapshot stays adopted). The e2e suite's read-modify-write helper pattern moves from `wh.policy.get()` to reading the harness-owned `policies.json` directly (`readPolicyFile()` — the file *is* the adopted policy there, since `setPolicy` fails unless the reload reports adoption). The policy document types (`Policy`, `TablePolicy`, `RolePermissions`) stay exported from the SDK — they describe `policies.json` and the e2e harness consumes them; `GET /v1/ops/pipes[/{name}]` is untouched. diff --git a/README.md b/README.md index e317a429..1914f7f2 100644 --- a/README.md +++ b/README.md @@ -130,7 +130,7 @@ go install github.com/Wave-RF/WaveHouse/cmd/wavehouse@latest `go install` compiles from source with cgo enabled (requires a C toolchain and glibc — Linux amd64/arm64 or macOS arm64) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse loads at start. Fetch it once before the first run: ```bash -go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch +go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch ``` This downloads 160–290 MB into the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision); point `WH_CHTYPES_REGISTRY` elsewhere if you keep it somewhere else. diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index b82f0ce7..a04d0127 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -367,7 +367,7 @@ The same policy drives every data path, but not every field is meaningful on eve :::caution[Live streams enforce column and row policy, but not resource limits] SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, ClickHouse's code 53; on an integer column such a claim is a `filter`), `decline` (the engine would not answer), `unavailable` (the tenant's ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list: a filter over a column the published row does not carry fails closed for that subscriber rather than being treated as a mismatch on an absent value. A filter naming a column the table no longer has, or a `MATERIALIZED`/`ALIAS`/`EPHEMERAL` one, fails to **compile**, which withholds every row for that role until the filter or the table changes — logged once, not per event. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, ClickHouse's code 53; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine cannot compile (a constant the column type cannot read) withholds every row for that role until the filter or the table changes — logged once, not per event. Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index dd27e108..90b98a38 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -268,9 +268,10 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is `invalid json` | | 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | -| 400 | `{"error":"missing dedupe id field \"event_id\""}` | **Per-record.** Only with `dedupe.require_id: true`, when the row carries no value for the configured `id_field` (an absent column or a `null` cell). With `require_id: false` (the default) the row is published un-deduped instead. Either way it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` | +| 400 | `{"error":"missing dedupe id field \"event_id\""}` | **Per-record.** Only with `dedupe.require_id: true`, when the row carries no value for the configured `id_field` (an absent column, a `null` cell or an empty string — the value an omitted `String` id column stores). With `require_id: false` (the default) the row is published un-deduped instead. Either way it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (the gate surfaces the token reason rather than silently falling back to `default_role`) | | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table — checked once, before any record | +| 403 | `{"error":"insert permissions were not resolved for this request"}` | The grant that resolved for the request is not an insert grant (an internal wiring fault, logged at `ERROR`). It is true for every record or none, so it fails the whole request — including an empty array, which answers `403` rather than `200` | | 403 | `{"error":"check failed for column \"x\""}` (several: `check failed for columns "x", "y"`) | **Per-record.** The row does not satisfy the role's insert [`check`](/access-control#insert-checks). The filter is AND-joined over every checked column, so with more than one it names the set that was tested rather than guessing an attribution | | 403 | `{"error":"policy check references column \"x\", which table \"t\" does not have"}` (also `… which is materialized and cannot be inserted`, the same for `alias`, and `… which is ephemeral and is never stored`) | **Per-record.** A **policy misconfiguration**, not a bad request: the role's `check` names a column the table lacks, one ClickHouse computes, or an `EPHEMERAL` one. None can be enforced, so the check would have passed silently while enforcing nothing. It fires on every insert by that role until the policy or the table is corrected, and names every offending column. `wavehouse validate` cannot catch it — it never sees the ClickHouse schema | | 404 | `{"error":"unknown table: ..."}` | Table not found in the tenant's discovered schema | @@ -284,7 +285,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure, for that tenant alone) or not open (see [Message Queue](/settings-directory#message-queue)). Under [`mq.backend: nats`](/deployment#external-nats), the table's partition stream is full, which refuses every table in it, or the tenant's table holds as many unwritten rows as the stream allows one subject. Response includes `Retry-After: 30` header. With dedupe on, the record's id is given back, so the retry is published rather than reported as a duplicate. | | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`, a transient broker failure, not a full queue). Only under [`mq.backend: nats`](/deployment#external-nats), including a partition stream the operator deleted; the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the record's id is left to lapse rather than given back, so a retry cannot land as a second copy; `Retry-After` is that lease, rounded up to whole seconds, when dedupe was on for the record, else the flat `Retry-After: 5`. | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | Dedupe is on and another request carrying the same id is still being published — usually a client's timeout-retry racing its own original. Its outcome decides whether this record is a duplicate, so retry after the `Retry-After` header (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). | -| 503 | `{"error":""}` | The tenant's ClickHouse line has no installed chtypes artifact for its server version, or its server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line). The body names the cause. Only that tenant is refused — every other tenant keeps working — and it is decided before the body is read, with `Retry-After: 5`. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) | +| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet (no completed schema refresh), a table could not be compiled, its ClickHouse line has no installed chtypes artifact for its server version, or its server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line). The body is generic on purpose — the cause, with zone names and artifact search paths, goes to the server log only. Only that tenant is refused — every other tenant keeps working — and it is decided before the body is read, with `Retry-After: 5`. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | **curl example:** @@ -298,7 +299,7 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ #### Timestamp rendering -WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling ClickHouse's own parser accepts under `date_time_input_format=best_effort` (the setting WaveHouse pins, both at chtypes' ingest compile and on the worker's `INSERT`) is accepted — RFC 3339 with any offset, a zone-less `YYYY-MM-DD[ T]HH:MM:SS[.fff]` read in the column's declared zone else the server's default, a Unix-seconds string, a bare integer at the column's tick scale, among the other forms its lenient parser reads. It is ClickHouse's grammar, not a reimplementation of it, so whatever a real `INSERT` into this table would accept, ingest accepts, with the same coercions and the same refusals. +WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling ClickHouse's own parser accepts under `date_time_input_format=best_effort` (the setting WaveHouse pins, both at chtypes' ingest compile and on the worker's `INSERT`) is accepted — RFC 3339 with any offset, a zone-less `YYYY-MM-DD[ T]HH:MM:SS[.fff]` read in the column's declared zone else the server's default, a Unix-seconds string, a bare integer (ClickHouse 26.8 reads a bare number in a `DateTime64` column as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — send milliseconds as a quoted string, or as a decimal number of seconds), among the other forms its lenient parser reads. It is ClickHouse's grammar, not a reimplementation of it, so whatever a real `INSERT` into this table would accept, ingest accepts, with the same coercions and the same refusals. **Outbound**, `DateTime`/`DateTime64` values in the NATS/SSE wire row and in `/v1/query` / `/v1/pipes/{name}` results are the exact bytes ClickHouse's writer produces, in the column's declared zone else the server's default: `"2026-06-21 04:00:00.123"`, space-separated, no `Z` suffix, never RFC 3339. Every consumer renders from the same stored value the same way, so SSE and `/v1/query` agree on spelling for a given row **by construction**, with no WaveHouse rewriting step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). The raw-SQL proxy `/v1/ops/query` is the exception: it sets `date_time_output_format=iso`, which keeps trailing fraction zeros and an ISO-8601 `Z`, and is not expected to match the other two byte-for-byte. @@ -360,7 +361,7 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ #### Batch Ingest -A JSON array, an NDJSON body, a CSV body or a TSV body ingests a batch in one request. Each record is validated, authorized, deduplicated and published independently, so **one malformed or rejected record never blocks the rest of the batch** — including inside a single-line (compact) JSON array, which WaveHouse re-frames in place before handing it over. An explicit empty array (`[]`) is a valid record-less batch (`200`, `total: 0`); blank lines in an NDJSON body are skipped. (The SDK's `insert([...])` array helper uses the NDJSON form automatically; every form returns the same response.) +A JSON array, an NDJSON body, a CSV body or a TSV body ingests a batch in one request. Each record is validated, authorized, deduplicated and published independently, so **one malformed or rejected record never blocks the rest of the batch** — including inside a single-line (compact) JSON array, which WaveHouse re-frames in place before handing it over. An explicit empty array (`[]`) is a valid record-less batch (`200`, `total: 0`) for a role whose insert grant resolves; blank lines in an NDJSON body are skipped. (The SDK's `insert([...])` array helper uses the NDJSON form automatically; every form returns the same response.) The response counts records read, published, rejected and deduplicated, then lists per-record outcomes: each entry mirrors the single-object response (`ok` / `duplicate` / `error`) plus its 1-based `index`, and carries ClickHouse's numeric `exception_code` when the rejection was its parser's. `results` is truncated to the first 10,000 entries; the four counts stay authoritative. @@ -401,6 +402,7 @@ A `200` is returned whenever the body was read and the records were processed | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (same auth gate as the single-object path; surfaces the token reason) | | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table (checked once, before any record) | +| 403 | `{"error":"insert permissions were not resolved for this request"}` | The grant that resolved for the request is not an insert grant; see the [single-record table](#post-v1ingesttabletable--ingest-data). Fails the whole request, an empty array included | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | | 500 | `{"error":"publish failed"}` / `{"error":"dedupe failed"}` | Message-queue or dedup-backend failure mid-batch, other than a full queue or an unreachable broker (below). After a publish failure the records before it keep their ids, so a whole-batch retry reports those as duplicates; the failing record's id is left to lapse as on the single-object path, and the rest of its window's ids are given back | @@ -408,7 +410,7 @@ A `200` is returned whenever the body was read and the records were processed | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`), mid-batch. Only under [`mq.backend: nats`](/deployment#external-nats); the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the failing record's id is left to lapse rather than given back — so `Retry-After` is that record's dedupe lease, rounded up to whole seconds, when it was deduped; a record published un-deduped has no lapsing claim to wait out, so `Retry-After: 5` | | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now; `Retry-After: 5`. Nothing in the window being reserved was published; the windows before it were, and keep their ids | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | A record's dedupe id is held by another request still being published; includes `Retry-After` (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). Nothing in that record's window was published; the windows before it were | -| 503 | `{"error":""}` | The tenant's ClickHouse line has no installed chtypes artifact, or its server time zone differs from the zone this process opened that line with — decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | +| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet, or its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the zone this process opened that line with — the cause is in the server log, not the body. Decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] @@ -664,7 +666,7 @@ Each SSE connection is bound to a single `?table=`; to consume multiple tables, Values of top-level `DateTime`/`DateTime64` columns inside `row` are ClickHouse's own rendering of the stored value — the exact bytes chtypes' `RowsExport` produced for that record (see [Timestamp rendering](#timestamp-rendering)), not a WaveHouse rewrite — so a live event and a `/v1/query` read of the same row agree on spelling **by construction**, with no separate canonicalization step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). A column declared with a non-UTC zone streams in that zone, not normalized to UTC; parse the timestamp with a zone-aware parser rather than assuming `Z`. -**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause — a connection is never delivered a row the query path would hide for that role, and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: its rows are withheld with reason `unavailable` while other tenants' streams are unaffected. The residual payload-vs-stored case is an event whose insert later fails outright at ClickHouse — a connectivity fault or batch error, not a data-shape problem chtypes would already have caught — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next live event (an in-flight gap-fill finishes under the policy snapshot taken when the stream opened), but an expired token or changed claims take effect only when the client reconnects. +**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause — a connection is never delivered a row the query path would hide for that role, and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: its rows are withheld with reason `unavailable` while other tenants' streams are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant's line is not served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert later fails outright at ClickHouse — a connectivity fault or batch error, not a data-shape problem chtypes would already have caught — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next live event (an in-flight gap-fill finishes under the policy snapshot taken when the stream opened), but an expired token or changed claims take effect only when the client reconnects. **CORS:** `/v1/stream` honors the request's tenant's `cors.allowed_origins` allowlist (settings directory) like every endpoint — the preflight included, which a browser sends without `X-Tenant-ID`, so over [a nested settings directory](/deployment#multi-tenant-deployments) the fronting proxy has to set the header on the `OPTIONS` too. Note that a **header-authenticated stream preflights before it connects** — `Authorization` is not CORS-safelisted — where a bare `EventSource` never preflighted at all: its request is not a `fetch()`, so Fetch's unsafe-request flag is never set and `Last-Event-ID` rides on the plain `GET`. Both headers are allow-listed, so an allowed origin connects *and* resumes cross-origin. @@ -890,7 +892,7 @@ The request that produced this envelope omitted `received_timestamp` (`DEFAULT n | `received_timestamp` | string | RFC 3339 nano timestamp when WaveHouse received the event. | | `format` | string | Row format. Always `JSONCompactEachRow` today; stated on the wire so a reader can tell an envelope it understands from one it doesn't. | | `columns` | string[] | The table's **wire** column names, in declaration order — what each position in `row` means (`internal/typelayer.Table.WireColumns`: the insertable subset minus any `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column — none of the three can be named in an `INSERT`, or is ever part of a published row). | -| `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order — the exact bytes ClickHouse's own writer produced for this stored row (`internal/typelayer.Table.Ingest`, via chtypes). A column the request body omitted carries its evaluated `DEFAULT` (or the type's implicit zero value where none is declared), not `null` — the same as a native `INSERT` naming fewer columns than the table has. `DateTime`/`DateTime64` values are ClickHouse's own rendering (see [Timestamp rendering](#timestamp-rendering)), and an out-of-range integer is wrapped the way a real `INSERT` wraps it. | +| `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order — the exact bytes ClickHouse's own writer produced for this stored row (`internal/typelayer`'s `Table.IngestWith`, via chtypes). A column the request body omitted carries its evaluated `DEFAULT` (or the type's implicit zero value where none is declared), not `null` — the same as a native `INSERT` naming fewer columns than the table has. `DateTime`/`DateTime64` values are ClickHouse's own rendering (see [Timestamp rendering](#timestamp-rendering)), and an out-of-range integer is wrapped the way a real `INSERT` wraps it. | `columns` and `row` are only meaningful together: a reader that cannot pair them — a length mismatch, an undecodable row, a `columns` list naming one column twice — has no way to map a value to a column. Both readers also refuse an envelope whose `format` they do not recognize. Either way the SSE fan-out withholds such an envelope rather than guess, and the batch consumer parks it on the DLQ with `X-DLQ-*` headers — acking and dropping it only where the DLQ is switched off for that table, since it can never insert on retry. Both outcomes increment `wavehouse_ingest_poison_total`, separated by its `disposition` label (`parked` / `dropped`). diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 7945b050..98ae5626 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -81,14 +81,14 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **router.go** — Route definitions. Public: `/livez`, `/readyz`, and the content-free `/v1/health` SDK ping (plus the permanent `/healthz` alias and the deprecated `/health`, `/ready` aliases). Policy-gated: `/v1/ingest?table={table}`, `/v1/query?table={table}` (structured), `/v1/pipes/{name}` (named pipes), `/v1/stream`. Admin-only (`RequireAdmin` — role == `policy.admin_role`, or a request bearing the operator key's operator bit, which passes even under a nil policy; over a nested settings directory `NewRouter` mounts the gate with no policy at all, whatever `Dependencies.PolicySource` was wired, so the operator key alone passes): `/v1/ops/schema/*`, `/v1/ops/dlq/stats`, `GET /v1/ops/pipes[/{name}]`, `/v1/ops/settings/reload`, `/v1/ops/query` (raw SQL — same gate as the rest of `/v1/ops/*`). `NewOpsRouter` is the router of a process without the `api` role: the probes and their aliases, `/version` and the same-port metrics path (the part it shares with `NewRouter`, `newProbeRouter`), and `POST /v1/ops/settings/reload` behind `RequireAdmin(nil)`, so only the operator key passes; every other route is a 404, under `/v1/ops` only once that gate has passed. - **auth middleware** — the JWT/JWKS authentication middleware is its own package, [`auth/`](#auth--authentication); the router runs it on every `/v1/*` route. - **tenant.go** — `TenantMW` resolves the request's tenant ahead of the auth middleware on every `/v1` route outside `/v1/ops/*`: the [`X-Tenant-ID`](/deployment#multi-tenant-deployments) header (absent means `tenant.Default`), validated by `tenant.Parse` (`400`), looked up in the `settings.Registry` (`resolveStore`: `404` for an id it does not hold, a bare `503` for a tenant whose folder was rejected — the findings stay out of a body answered before authentication), and the resolved `*settings.Store` stored in the request context (`WithStore` / `StoreFromContext` — here rather than in `tenant/`, because `settings` names `tenant.ID`). A handler reads the store once and passes it down as an argument — the per-tenant getters it holds take it as a parameter (`(*settings.Store).Policy`, `.DedupeFor`, `.DefaultMaxRows`, … in production) — and nothing below a handler reads the context; a tenant route reached without a resolved store answers `500` rather than fall back to a tenant. The probes, `/version`, the metrics path, and `/v1/ops/*` are tenant-exempt; an ops route that addresses one tenant — the admin pipe reads, the schema routes, the raw-SQL proxy, the settings reload and the DLQ stats — names it in `?tenant=` (`opsTenant`, and `opsStore` over it for the routes that need the tenant's store; the DLQ stats need none, since the MQ holds the queue), parsed strictly so that a query `url.ParseQuery` would half-read is a `400` rather than a read of the default tenant, which is what it means when absent on the reads (on the reload, absent is the whole directory). -- **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`clickhouse_exec.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. +- **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`sql_classify.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. -- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch, and the dedupe id is read positionally out of the exported row. One `typelayer.Table.Ingest` call per body, on the table set bound for the request's tenant, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5`, decided before the body is read and ahead of every other tenant's traffic. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent or `null` can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). +- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch, and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read and ahead of every other tenant's traffic. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` — a deliberately different audience from the structured-query path, which leaves ClickHouse's default spelling alone so it matches the SSE wire. - **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every filter value bound as a named `{pN:String}` parameter, so ClickHouse renders each row and WaveHouse only frames the lines into an array. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=simple`) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. Connections per tenant are capped at the tenant's `max_open_conns`. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). -- **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) driver.Conn` beside it — each resolves the request's tenant per call, and a nil connection (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). +- **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) chconn.Target` beside it — each resolves the request's tenant per call, and the zero target (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for one tenant's per-table parked counts (optionally one table) and its total — the tenant `?tenant=` names, read strictly by `opsTenant`, tenant `0` without it. The tenant is looked up in the MQ, not the settings registry, so a rejected or removed tenant's parked rows are read like a served one's; a tenant with no dead-letter queue (`mq.ErrNoDeadLetterQueue`) is a 404, and any other failure to read it a 500. The queue itself is `internal/mq`'s. - **health.go** — Liveness (`/livez`), readiness (`/readyz`), and a content-free `Online` ping (`/v1/health`, the SDK's public liveness check); `/healthz` is a permanent alias of `/livez`, and `/health`/`/ready` are deprecated aliases. All three consult an optional `BootState` so they can return 503 while boot-time schema discovery is still failing in the retry loop (see `internal/app`; over a nested directory, while no tenant's has succeeded); once `BootState.Set(nil)` fires, `/livez` returns 200 and stays there. `/readyz` additionally runs a `Ping` each call — `chconn.Pools.Ping` in production: every open pool at once, ready at the first answer, every pool's error joined when none answers; `/v1/health` deliberately does not. @@ -167,10 +167,10 @@ The only package that imports `github.com/wave-rf/chtypes/go/chtypes` (the sole Each tenant has its own table set inside that engine, bound from the tenant's own schema refresh and released when the tenant is no longer served. -- **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind. A tenant whose ClickHouse line has no installed artifact is unavailable on its own, and so is one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Either way the cause is named, every other tenant keeps working, and a later bind of the same tenant (a reload, a refresh that now agrees) clears it. -- **`Engine.RoleTable`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column is simply not in the schema, so naming it is ClickHouse's code 117, and an absent check column takes the claim as its default while a supplied value still wins. -- **`Table.Ingest(format, body)`** runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. -- **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `Table.Ingest` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow` / `Row.Visible` judge a subscriber's row filter over one parsed event, with compiled filters cached per table. Only a definite true admits; a predicate error, a policy column the table no longer has, schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. +- **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind. A tenant whose ClickHouse line has no installed artifact is unavailable on its own, and so is one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Either way the cause is recorded (an ingest client sees only a generic `503`; the cause goes to the log), every other tenant keeps working, and a later bind of the same tenant (a reload, a refresh that now agrees) clears it. +- **`Engine.RoleTable(tenant, table, shape)`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column is simply not in the schema, so naming it is ClickHouse's code 117, and an absent check column takes the claim as its default while a supplied value still wins. +- **`Table.IngestWith(format, opts, body, checks...)`** (`Ingest` is the same with no options) runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. +- **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per table. Only a definite true admits; a predicate error, a policy column the table no longer has, schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. - A tenant that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row for that tenant's tables with reason `unavailable`. See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest error-response shape and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced) for how predicates are compiled and evaluated. @@ -179,7 +179,7 @@ See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest e - **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line — the exact bytes ClickHouse's own writer produced for the stored record — with `columns` naming its positions (the table's insertable columns, or the narrower list a column-restricted role produced); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with `typelayer.InsertSettings()` (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) plus `async_insert=0`, the same parsing settings chtypes compiled the row with. The worker never loads the artifact: `InsertSettings` is static. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so under `mq.backend: embedded` the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Under `mq.backend: nats`, anyone the operator lets publish to `.ingest.>` reaches it too, past auth, policy and schema validation, so that right belongs to the `wavehouse` user alone. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy — or through an operator-authored [pipe that writes](/pipes#pipes-that-write), gated only by its `allowed_roles`. A batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling) is never tried: no row of it could pass, so `parkBatch` takes it to the DLQ switch whole, logging once per batch rather than twice per row. Otherwise a bulk-insert failure is first classed by `chconn.Classify`: a ClickHouse that cannot take the insert (unavailable, denied, or no verdict at all) sends the batch back to the MQ for a delayed redelivery (`retryLater` → `mq.Message.NakWithDelay`), under a backoff shared by every table on the same pool (a failure of one table — read-only, too many parts — backs off that table alone), and never to the DLQ — the same when it stops answering mid-isolation. Only when ClickHouse rejects the batch, or refuses a multi-row batch for its size (`chconn.Splittable`: too many partitions for one INSERT, the memory limit), is it re-inserted row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. - **backoff.go** — The retry backoff behind `retryLater`: a small circuit breaker per ClickHouse pool (the target's URL, user and database), and one per pool and table for a failure of one table (`chconn.TableScoped`). A failure opens it for 1 s, doubling to a 30 s cap, each window jittered down to half; while it is open, flushes and arriving rows are handed back without a request, and once it elapses one flush probes. Any answer that is not an outage closes it. -- **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. `Row` is the exact bytes `internal/typelayer.Table.Ingest` returned for an accepted record — ClickHouse's own `JSONCompactEachRow` writer output, `DEFAULT`s already filled in — not a value WaveHouse encodes itself. +- **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. `Row` is the exact bytes `internal/typelayer`'s `IngestWith` returned for an accepted record — ClickHouse's own `JSONCompactEachRow` writer output, `DEFAULT`s already filled in — not a value WaveHouse encodes itself. - **claims.go**, **assign.go** — `ClaimShards` wraps a `Sharded` queue so an ingest process consumes only its share of the units, one process per unit at a time. Each process holds one membership lease, `ingest.m` for the lowest free `j` under the number of configured units (at most 64, read 16 at a time each tick), and every tick (2s) reads which slots are live (`coord.Observer.Held`); `assignUnits` gives every unit to a live slot by rendezvous hashing capped at ⌈units/slots⌉, the same result in every process for the same view; the extras are assigned apart from the configured units, so processes whose lists of extras differ (each refreshes it every five minutes) still agree on every configured unit. A unit no longer assigned halts (`mq.Halter`: stops fetching, keeps its pin, returns once what it fetched reached the worker), waits (bounded by the worker's 60s ack wait, the shortest `ack_wait` a durable may have) for the rows it delivered to settle (`Message.OnSettled`), then releases its pin (`mq.Releaser`). `Halt` on the claiming consumer ends the ticks and halts every unit at once; the worker calls it before its final flush, and the claims' stop releases the units and resigns the membership lease after it. A unit taken over from a slot that is no longer live is reset (`ResetOrphaned`) before it is bound — waiting, unbound, while the broker answers `ErrUnitHeld` (the dead owner's pin has not lapsed, or the unit was active within its pinned TTL), up to 30s — so the dead owner's unacked rows come back at once; on a process's first tick only if it is the one live member (a restart after a clean stop; after a crash the dead run's lease still counts as live for a lease duration). Each unit's rows delivered and unsettled are capped at its share of the worker's 10,000 (`ClaimConfig.MaxHeld`): an even share over the units assigned (`unitShare`, recomputed each tick and read by the broker before every fetch through `ConsumerConfig.MaxHeld`), never under 1,000, two batches; at its share a unit fetches only what keeps its pin, and one stuck unit never takes another's. A row stops counting at its first ack or nak attempt (`Message.OnSettled` fires whether or not the broker confirms), or after `ack_wait` if the worker never settles it, since the broker redelivers it then as a new row. A configured unit whose delivery ends, or whose durable or stream is gone or whose durable no longer fits (`mq.ErrConsumerNotFound`, `mq.ErrConsumerMismatch`) when it is bound, fails the worker; any other bind failure is retried next tick, logged as an error once it has lasted a minute, and an extra's end is logged. The lowest live slot counts the units with rows and no owner (`Sharded.Unowned`) for `wavehouse_ingest_shards_unowned`. - **sweeper.go** — `Sweeper` implements the Active Sweeper pattern. It runs every minute and asks the MQ (`mq.Purger.PurgeAcked`) to drop the ingest events that are **both** ACKed by the buffer consumer (written to ClickHouse) **and** older than the gap window (re-read every sweep: each tenant's own `stream.gap_window_minutes`, a rejected tenant's as its folder last had it (unbounded for one rejected since boot) — `internal/app`'s `gapWindows` — and none for a removed tenant). Finding the purge point is `internal/mq`'s (`purge.go`). @@ -214,7 +214,7 @@ The package's design invariants — stdout always 100%, WARN+ERROR always export - **canonical.go** — the one rendering layer for policy comparison operands: every JWT claim value (`CanonicalScalar`) is rendered into one exact canonical decimal form (positional, digit-bounded, never a float64 round-trip) before it reaches a `filter`/`check` predicate, so every comparison surface binds the same value the same way, and a null/object/array claim fails closed. A policy-authored literal is *not* re-rendered — it binds exactly as written, and a spelling the column cannot read is ClickHouse's own type error at evaluation time on both surfaces. On an integer column the bound claim is additionally compared through the strict round-trip cast (`WhereSQL(colType)` renders the predicate for the query builder; the stream's chtypes filter uses the same shape). - **source.go** — `Source`, a `func() *Policy` the `/v1/ops` gate (over a flat settings directory) reads per call, so a settings reload applies to the very next request; in production it is the default tenant's `settings.Store.Policy`, and `Static(p)` fixes one for tests. The tenant-aware surfaces take a keyed variant that resolves to the same `Store.Policy`: `api.PolicySource` (`func(*settings.Store) *policy.Policy`) for ingest, structured query and pipes, `stream.PolicySource` (`func(tenant.ID) *policy.Policy`) for the hub, and `auth.PolicySource` (same shape) for the operator key's admin role. A `nil` result is a deliberate lockout. -Predicate *evaluation* lives elsewhere: `internal/typelayer` compiles a role's resolved predicates through chtypes and evaluates them — `Table.ParseRow` / `Row.Visible` for a streamed event, `Table.Ingest` (a row filter inside the validating parse) for an ingest `check` — with ClickHouse's own comparison semantics for every column type. `policy` only resolves the values both the SQL path and that engine bind. +Predicate *evaluation* lives elsewhere: `internal/typelayer` compiles a role's resolved predicates through chtypes and evaluates them — `Table.ParseRow` / `Row.Visible` for a streamed event, `Table.IngestWith` (a row filter inside the validating parse) for an ingest `check` — with ClickHouse's own comparison semantics for every column type. `policy` only resolves the values both the SQL path and that engine bind. ### `pipes/` — Named Query Pipes diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index 27740d2f..f410be44 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -170,6 +170,16 @@ Only the secret and the connection ceiling are boot config. The wiring — nativ | `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the `max_open_conns` of the open pools — one per distinct connection tuple among the served tenants, see [the settings directory](/settings-directory#clickhouse) — must not add up to more. Pools above it at boot refuse to start, naming the sum and the ceiling; on a reload a pool resized above it is refused and keeps its size, and a new pool that would cross it is not opened — its tenants keep the pool they had, or have none — both logged (the reload itself still reports `adopted`) and retried by the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. | | `clickhouse.chtypes_registry` | `WH_CHTYPES_REGISTRY` | *(empty)* | Directory holding the chtypes artifacts (one `/` per ClickHouse line). Empty defers to the chtypes search path — `$CHTYPES_REGISTRY`, `~/.cache/chtypes/artifacts/abi6/-`, then the system directories. Either way a library is opened lazily, on first use of its line; an explicit directory is searched first, then the rest of the path. Boot-tier: changing it is a restart. See [chtypes artifacts](/deployment#chtypes-artifacts). | +### ClickHouse user access + +The ClickHouse user WaveHouse connects as (the settings directory's `clickhouse.username`) needs these, and nothing about the tables themselves has to be exposed to anyone else: + +- **`SELECT` on the tenant's tables.** Structured queries, read pipes and the stream's replay read through it. +- **`INSERT` on the tenant's tables, for the ingest worker.** The worker writes each batch with `INSERT … FORMAT JSONCompactEachRow` over the HTTP interface. A process that only serves reads never inserts, but the same user is used by every process, so grant it wherever any process ingests. +- **Read access to `system.columns` and `system.tables`**, plus `SELECT timezone()` and `SELECT version()`, which need no grant. Schema discovery reads the columns, their `DEFAULT`/`MATERIALIZED` kinds and each table's DDL, and records the server's version and time zone; the version picks the [chtypes artifact](/deployment#chtypes-artifacts) and the zone is the one the in-process parser reads bare timestamps in. ClickHouse lists only the tables a user holds a grant on, so a table the user cannot see is a table WaveHouse does not know. +- **A profile that lets it change per-query settings.** WaveHouse sends settings with every query: reads carry `readonly=2` together with `max_execution_time`, `wait_end_of_query`, `http_write_exception_in_output_format`, `cancel_http_readonly_queries_on_client_close` and the JSON/date rendering settings, and the ingest worker's `INSERT` carries `date_time_input_format`, `input_format_null_as_default` and `async_insert=0`. A profile with `readonly=1` (or a `` entry that pins any of these) makes ClickHouse refuse them: a read answers `502 clickhouse.misconfigured` (the `READONLY` refusal), and the worker's batches fail. Use `readonly=0`, or `readonly=2` for a user that only ever reads: `readonly=2` still forbids the ingest worker's `INSERT`. +- **Whatever else the statements you hand it need.** A [pipe that writes](/pipes#pipes-that-write) and the admin-only raw SQL endpoint `/v1/ops/query` run as this user too, so a pipe that runs `ALTER` needs the grant for it, and a missing grant is `403 clickhouse.access_denied`. + ### Server-side resource limits WaveHouse enforces a role's **per-role** resource caps (from the [access-control policy](/access-control#resource-limits)) by attaching them to each query as ClickHouse settings. **Server-wide** limits — the backstop that applies to *every* query regardless of role, including raw admin SQL — are configured in **ClickHouse itself**, via its [settings profiles](https://clickhouse.com/docs/operations/settings/settings-profiles) and [quotas](https://clickhouse.com/docs/operations/quotas). This keeps one authoritative place for global governance, has ClickHouse enforce it natively (defense-in-depth, even against a WaveHouse bug), and lets you use standard ClickHouse operations. diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index b1e02c80..d7a191b9 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -106,11 +106,13 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ ## chtypes artifacts -**What it is.** WaveHouse validates ingest data, coerces types, substitutes `DEFAULT`s, and evaluates row-level security by running ClickHouse's own parser in-process, through [chtypes](https://github.com/wave-rf/chtypes) (`github.com/wave-rf/chtypes/go` v0.4.0). The parser itself ships as a shared library (`libchtypes.so` / `.dylib`) built per **ClickHouse minor line** (e.g. `26.6`) and per platform. There is no nearest-version fallback: each tenant's table set is compiled from its own schema refresh against the artifact matching that tenant's ClickHouse server, and a library is opened lazily, the first time a tenant on that line is bound. +**What it is.** WaveHouse validates ingest data, coerces types, substitutes `DEFAULT`s, and evaluates row-level security by running ClickHouse's own parser in-process, through [chtypes](https://github.com/wave-rf/chtypes) (`github.com/wave-rf/chtypes/go` v0.5.1). The parser itself ships as a shared library (`libchtypes.so` / `.dylib`) built per **ClickHouse minor line** (e.g. `26.8`) and per platform. There is no nearest-version fallback: each tenant's table set is compiled from its own schema refresh against the artifact matching that tenant's ClickHouse server, and a library is opened lazily, the first time a tenant on that line is bound. + +**The line this build is tested on.** The repository pins ClickHouse **26.8** (`chtypes.lock` names the 26.8.15.10 build, and the compose files and the integration suite run `clickhouse/clickhouse-server:26.8.15.10`). The type layer inherits ClickHouse's own parsing rules, so they are the connected server's: for instance, 26.8 reads a bare JSON number in a `DateTime64` column as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — WaveHouse stores what the server would. **Which processes load it.** Only processes with the `api` role — the ones that serve ingest and the stream. A process that runs only the ingest worker or the sweeper loads no artifact and boots without one installed. An API process refuses to start when no artifact is installed at all. -**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with a message naming both zones; run such tenants in a separate process, or align the servers' `timezone` setting. See [API → Ingest error responses](/api#error-responses) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). +**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with the cause in the server log; run such tenants in a separate process, or align the servers' `timezone` setting. See [API → Ingest error responses](/api#error-responses) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). **Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse does not autofetch on a miss in production — an unmatched line is a boot-time or refresh-time failure, not a background download. @@ -121,20 +123,20 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ **Release archives and `go install` / building from source** do not carry or fetch an artifact — only the Docker images bake one in. See the [README's `go install` caveat](https://github.com/Wave-RF/WaveHouse#c-go-install-binary-no-docker). Fetch one yourself before first run: ```bash -scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch --frozen --lock chtypes.lock 26.6 +scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch --frozen --lock chtypes.lock 26.8 ``` or, for a line not in the repo's lock file: ```bash -go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch +go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch ``` ### Pinning with `chtypes.lock` -`chtypes.lock`, checked in at the repo root, records the exact artifact file and SHA-256 per platform/line the project builds and tests against. CI restores from it with `--frozen` (refusing anything the lock doesn't name) rather than fetching the rolling artifact release, so a pipeline never silently starts testing a new build. Refresh it deliberately — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch --lock chtypes.lock --platform `, once per platform (`darwin-arm64`, `linux-amd64`, `linux-arm64`), without `--frozen` — and commit the result; don't regenerate it implicitly. +`chtypes.lock`, checked in at the repo root, records the exact artifact file and SHA-256 per platform/line the project builds and tests against. CI restores from it with `--frozen` (refusing anything the lock doesn't name) rather than fetching the rolling artifact release, so a pipeline never silently starts testing a new build. Refresh it deliberately — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch --lock chtypes.lock --platform `, once per platform (`darwin-arm64`, `linux-amd64`, `linux-arm64`), without `--frozen` — and commit the result; don't regenerate it implicitly. -A lock is specific to the SDK's ABI revision (6 at v0.4.0): the fetcher never selects a build from another revision, so after an SDK bump that changes the revision, `--frozen` fails (`CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED`) until the lock is regenerated the same way, and the CI cache key and path (`abi6`) move with it. +A lock is specific to the SDK's ABI revision (6 at v0.5.1): the fetcher never selects a build from another revision, so after an SDK bump that changes the revision, `--frozen` fails (`CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED`) until the lock is regenerated the same way, and the CI cache key and path (`abi6`) move with it. ## Releases diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index a33b205d..bd4bef27 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -26,7 +26,7 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: `internal/typelayer` loads a per-ClickHouse-version shared library at start to run ingest validation and row-level security through ClickHouse's own parser (see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)). It is not source code and `make tools` does not fetch it for you — pull it once with: ```bash -scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.4.0 fetch --frozen --lock chtypes.lock 26.6 +scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch --frozen --lock chtypes.lock 26.8 ``` It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–290 MB — expect the first run to take a minute or two. Without it, `make dev` / `make test` / `make test-e2e` fail closed (a `503` on ingest, every stream row withheld) until a matching artifact exists for the ClickHouse line the tests or your local server run against. diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index d062fb43..beccb4a4 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -59,7 +59,7 @@ flowchart LR Note the embedded broker's stream is **dual-use**: it is both the durable buffer feeding the worker and the replay buffer that SSE clients gap-fill from. That is why a custom sweeper exists there instead of plain work-queue auto-deletion; `mq.backend: nats` splits the two roles instead, with work-queue partitions and a separate history stream (see [Scaling out](#scaling-to-multiple-instances)). :::note[Omitted columns take their real DEFAULT, not `null`] -The batch that reaches `insertToClickHouse` is not assembled from the request body — it is the bytes `Table.Ingest` returned for each accepted record, produced by ClickHouse's own writer. An omitted field's `DEFAULT` (or the type's implicit zero) was evaluated before that line existed, so a `Nullable(T) DEFAULT …` column takes its default exactly as an `INSERT` naming fewer columns would. Verified on ClickHouse 26.6.3. +The batch that reaches `insertToClickHouse` is not assembled from the request body — it is the bytes `IngestWith` returned for each accepted record, produced by ClickHouse's own writer. An omitted field's `DEFAULT` (or the type's implicit zero) was evaluated before that line existed, so a `Nullable(T) DEFAULT …` column takes its default exactly as an `INSERT` naming fewer columns would. Verified on ClickHouse 26.8. ::: :::note[Insert settings pinned] diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index f17f64fc..bc9b319e 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -40,7 +40,7 @@ The SDK **never throws** for anything the server returns — all API errors come | 502 | `clickhouse.misconfigured` | No | ClickHouse refused WaveHouse's own credentials or database, or the route to it is wrong (a redirect, or a `4xx` other than `408`/`413`/`429`, with no exception code) — an operator fix | | 502 | `clickhouse.response_too_large` | No | A raw-SQL (`wh.sql`) response over the 64 MiB cap | | 503 | `clickhouse.unavailable` | Yes | ClickHouse is down, unreachable or overloaded; `Retry-After: 5`, honored between attempts | -| 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, a schema not discovered yet, a tenant on no ClickHouse pool, a dedupe store that cannot answer (`dedupe store unavailable`, `Retry-After: 5`), a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`), or a record whose dedupe id another request is still publishing (`a request with the same dedupe id is in flight`, `Retry-After`: the server's dedupe lease, 30 s by default). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on those last two causes waits that long; a stream re-dials on its own jittered backoff instead | +| 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, a schema not discovered yet, a tenant on no ClickHouse pool, a tenant whose ClickHouse line the type layer cannot serve (`ingest validation is unavailable`, `Retry-After: 5`), a dedupe store that cannot answer (`dedupe store unavailable`, `Retry-After: 5`), a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`), or a record whose dedupe id another request is still publishing (`a request with the same dedupe id is in flight`, `Retry-After`: the server's dedupe lease, 30 s by default). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on those last two causes waits that long; a stream re-dials on its own jittered backoff instead | | 0 | `NETWORK_ERROR` | Yes | Network failure (retried with exponential backoff) | | 0 | `ABORTED` | No | Request canceled via `AbortSignal` | | 0 | `SSE_CONNECT_ERROR` | No | Stream could not be started (e.g. a non-absolute `baseURL`) | From c6b0128616e4b6fd39127faac95fc86825364ab0 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:40:18 -0400 Subject: [PATCH 21/70] fix(query): parse timestamp filters in ClickHouse, send in lists as tables A filter value on a DateTime column was rewritten in Go to zone-less UTC text and bound as a String, which ClickHouse reads in the column's zone: on a DateTime('Asia/Tokyo') column `eq 2026-06-21T04:00:00Z` matched nothing, on DateTime64(3, 'America/New_York') the row four hours later, and on a zone-less column under a Europe/Berlin server the row two hours earlier (measured on 26.8.15.10 and 24.8.14.39). time_range bounds had the same shift. The builder no longer rewrites a value. A filter on a Date or DateTime column binds as {pN:String} and expands to parseDateTime64BestEffort({pN:String}, 8, '') (toDate or toDate32 of it for a date): an offset or Z gives the exact instant, a zone-less value still reads in the column's zone, a fraction compares exactly against a DateTime column, the primary key is used, and a value ClickHouse cannot parse is its own refusal (code 41, a 400). time_range bounds are RFC 3339 in UTC through the same parse. An in list now travels as a ClickHouse external table, read by `IN (SELECT (v) FROM _pN)`, instead of one Array(String) query parameter. A query parameter is capped at 128 KiB and the request line at 1 MiB; a table is read whole, so the 1 MiB request body is the only bound on a list, and the pre-check that answered 400 for a long one is gone. The subquery's set uses the primary key, where `IN arrayMap(...)` read every granule on 26.8 and was refused on 24.8. Elements are cast to the column's type with accurateCastOrNull, since a set of Strings compared the column as text (Decimal 3.00 missed '3.00'). A request with an in list is multipart: the statement in the query form field, the tables as file parts, scalars still on the query string. The remaining limits, each a 400 naming it: one scalar over 128 KiB encoded, all of them over the request line, and with an in list a statement over the 128 KiB form field. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/api/clickhouse_http.go | 117 ++++-- internal/api/clickhouse_http_test.go | 140 +++++-- internal/api/structured_query.go | 17 +- .../api/structured_query_integration_test.go | 327 +++++++++++++++ internal/api/structured_query_test.go | 59 ++- internal/query/bind.go | 374 ++++++++++++++++++ internal/query/builder.go | 269 ++----------- internal/query/builder_test.go | 357 +++++++++++------ 8 files changed, 1225 insertions(+), 435 deletions(-) create mode 100644 internal/api/structured_query_integration_test.go create mode 100644 internal/query/bind.go diff --git a/internal/api/clickhouse_http.go b/internal/api/clickhouse_http.go index 1a0b82c8..76998330 100644 --- a/internal/api/clickhouse_http.go +++ b/internal/api/clickhouse_http.go @@ -6,14 +6,15 @@ import ( "crypto/tls" "fmt" "io" + "mime/multipart" "net/http" "net/url" - "strconv" "strings" "sync" "time" "github.com/Wave-RF/WaveHouse/internal/chconn" + "github.com/Wave-RF/WaveHouse/internal/query" "github.com/Wave-RF/WaveHouse/internal/settings" ) @@ -60,13 +61,15 @@ var chReadSettingsFixed = map[string]string{ "output_format_json_named_tuples_as_objects": "1", } -// ClickHouse's own limits on the HTTP interface's query string, at their -// defaults: http_max_field_value_size per field and http_max_uri_size for the -// whole request line. Both count the percent-encoded text. Measured on -// 24.8.14.39 and 26.6.3.62: a 131072-byte value is read and a 132096-byte one -// is refused with code 1000 ("Field value too long"), and a 1 MiB value with -// a codeless 400 — answers chFailureOf would read as an outage and a -// misconfiguration rather than as a request that is too large. +// ClickHouse's own limits on the HTTP interface's fields, at their defaults: +// http_max_field_value_size per field — a query-string parameter, counted +// percent-encoded, or a multipart form field — and http_max_uri_size for the +// whole request line. Measured on 24.8.14.39 and 26.8.15.10: a 131072-byte +// query-string value is read and a 131073-byte one is refused with code 1000 +// ("Field value too long"), as is a 200 KiB form field, and a 1 MiB value +// with a codeless 400 — answers chFailureOf would read as an outage and a +// misconfiguration rather than as a request that is too large. An external +// table has no such limit: a 1.2 MiB one was read whole. const ( chMaxFieldBytes = 128 << 10 chMaxURIBytes = 1 << 20 @@ -82,10 +85,15 @@ const defaultReadConns = 100 // chRequest is one statement for ClickHouse's HTTP interface. type chRequest struct { sql string - // params supply param_p0 … param_pN-1, positionally, for the {pN:…} - // placeholders query.BuildResult.NamedParams emitted, already encoded - // for ClickHouse's parameter reader. - params []string + // params supply param_ for the {:String} placeholders + // query.BuildResult.Bind emitted, already encoded for ClickHouse's + // parameter reader. They ride on the query string. + params []query.Param + // tables are the external tables the statement's `in` lists read. With + // any, the request is multipart/form-data: the statement moves from the + // body into the query form field, one field per table describes it, and + // each table is a file part. + tables []query.Table // settings are per-query ClickHouse settings (chReadSettings). settings map[string]string // write runs the statement without readonly=2: a write pipe. Everything @@ -184,12 +192,16 @@ func (c *chReader) do(ctx context.Context, target chconn.Target, conns int, req for k, v := range req.settings { q.Set(k, v) } - for i, p := range req.params { - q.Set("param_p"+strconv.Itoa(i), p) + for _, p := range req.params { + q.Set("param_"+p.Name, p.Value) } u.RawQuery = q.Encode() - httpReq, err := http.NewRequestWithContext(ctx, http.MethodPost, u.String(), strings.NewReader(req.sql)) + reqBody, contentType, err := chRequestBody(req) + if err != nil { + return nil, fmt.Errorf("clickhouse request: %w", err) + } + httpReq, err := http.NewRequestWithContext(ctx, http.MethodPost, u.String(), reqBody) if err != nil { return nil, fmt.Errorf("clickhouse request: %w", err) } @@ -198,7 +210,7 @@ func (c *chReader) do(ctx context.Context, target chconn.Target, conns int, req for name, value := range target.Headers { httpReq.Header.Set(name, value) } - httpReq.Header.Set("Content-Type", "text/plain; charset=utf-8") + httpReq.Header.Set("Content-Type", contentType) if target.Username != "" { httpReq.Header.Set("X-ClickHouse-User", target.Username) } @@ -243,6 +255,41 @@ func (c *chReader) do(ctx context.Context, target chconn.Target, conns int, req return rows, nil } +// chRequestBody is req's body and its content type: the statement as plain +// text, or — when it reads external tables — a multipart form carrying the +// statement, each table's structure and format, and then the tables. The +// descriptions go first because ClickHouse reads a table's part as it +// arrives, with what it has been told about it so far. +func chRequestBody(req chRequest) (io.Reader, string, error) { + if len(req.tables) == 0 { + return strings.NewReader(req.sql), "text/plain; charset=utf-8", nil + } + var buf bytes.Buffer + w := multipart.NewWriter(&buf) + fields := [][2]string{{"query", req.sql}} + for _, t := range req.tables { + fields = append(fields, [2]string{t.Name + "_structure", query.TableStructure}, [2]string{t.Name + "_format", query.TableFormat}) + } + for _, f := range fields { + if err := w.WriteField(f[0], f[1]); err != nil { + return nil, "", err + } + } + for _, t := range req.tables { + part, err := w.CreateFormFile(t.Name, t.Name) + if err != nil { + return nil, "", err + } + if _, err := part.Write(t.Data); err != nil { + return nil, "", err + } + } + if err := w.Close(); err != nil { + return nil, "", err + } + return &buf, w.FormDataContentType(), nil +} + // jsonEachRowArray frames ClickHouse's newline-delimited JSONEachRow output as // the JSON array the read endpoints return, copying the rows through // untouched. A JSONEachRow row is one JSON object on one line — a newline in @@ -273,25 +320,45 @@ func jsonEachRowArray(body []byte) (rows, tail []byte) { return append(out, ']'), nil } -// checkParamSizes refuses bound values the HTTP interface would refuse: each -// one past ClickHouse's per-field limit, or all of them past what the request -// line holds. A long `in` list is the usual cause, and the caller is the one -// who can split it. -func checkParamSizes(params []string) error { +// checkRequestSize refuses a bound query the HTTP interface would refuse, +// at ClickHouse's own limits, so the caller gets a 400 naming the limit +// rather than code 1000 classed as an outage. A scalar value rides on the +// query string, each one capped and all of them together bounded by the +// request line; with an `in` list the statement rides in one form field. +// An `in` list itself has no cap: its table is read whole. +func checkRequestSize(b *query.Bound) error { total := 0 - for _, p := range params { - n := len(url.QueryEscape(p)) + for _, p := range b.Params { + n := len(url.QueryEscape(p.Value)) if n > chMaxFieldBytes { - return fmt.Errorf("filter value too large: %d bytes once encoded, over the %d ClickHouse's HTTP interface takes; split a long in list across queries", n, chMaxFieldBytes) + return fmt.Errorf("filter value too large: %d bytes once encoded, over the %d ClickHouse's HTTP interface takes for one value; an in list has no such limit", n, chMaxFieldBytes) } total += n } if total > chMaxURIBytes-chURIHeadroom { - return fmt.Errorf("filter values too large: %d bytes once encoded, over the %d ClickHouse's HTTP interface takes; split long in lists across queries", total, chMaxURIBytes-chURIHeadroom) + return fmt.Errorf("filter values too large: %d bytes once encoded, over the %d ClickHouse's HTTP interface takes for all of them; an in list has no such limit", total, chMaxURIBytes-chURIHeadroom) + } + if len(b.Tables) > 0 && len(b.SQL) > chMaxFieldBytes { + return fmt.Errorf("query too large: %d bytes of SQL, over the %d ClickHouse's HTTP interface takes in the form field a query with an in list travels in; use fewer filters", len(b.SQL), chMaxFieldBytes) } return nil } +// cacheValues are b's bound values as the cache key takes them: each +// parameter's value, then each table's bytes, exactly as they go on the wire. +// The statement, hashed alongside, names every parameter and table, so the +// split between the two is never ambiguous. +func cacheValues(b *query.Bound) []string { + out := make([]string, 0, len(b.Params)+len(b.Tables)) + for _, p := range b.Params { + out = append(out, p.Value) + } + for _, t := range b.Tables { + out = append(out, string(t.Data)) + } + return out +} + // targetOf is target's answer for store — the tenant's HTTP wiring — and the // zero Target, a tenant on no pool, for an unwired source. func targetOf(target func(*settings.Store) chconn.Target, store *settings.Store) chconn.Target { diff --git a/internal/api/clickhouse_http_test.go b/internal/api/clickhouse_http_test.go index 66b64649..dce0ca84 100644 --- a/internal/api/clickhouse_http_test.go +++ b/internal/api/clickhouse_http_test.go @@ -6,9 +6,12 @@ import ( "errors" "fmt" "io" + "mime" + "mime/multipart" "net/http" "net/http/httptest" "net/url" + "slices" "strconv" "strings" "sync" @@ -20,6 +23,7 @@ import ( "github.com/stretchr/testify/require" "github.com/Wave-RF/WaveHouse/internal/chconn" + "github.com/Wave-RF/WaveHouse/internal/query" "github.com/Wave-RF/WaveHouse/internal/settings" ) @@ -46,12 +50,35 @@ type chSeen struct { sql string query url.Values header http.Header + // fields and files are a multipart body's form fields and file parts, in + // the order they arrived; sql is then its query field. + fields [][2]string + files []query.Table } func (f *fakeCH) RoundTrip(r *http.Request) (*http.Response, error) { body, _ := io.ReadAll(r.Body) _ = r.Body.Close() seen := &chSeen{sql: string(body), query: r.URL.Query(), header: r.Header.Clone()} + if mt, params, err := mime.ParseMediaType(r.Header.Get("Content-Type")); err == nil && mt == "multipart/form-data" { + seen.sql = "" + mr := multipart.NewReader(strings.NewReader(string(body)), params["boundary"]) + for { + part, err := mr.NextPart() + if err != nil { + break + } + data, _ := io.ReadAll(part) + if part.FileName() != "" { + seen.files = append(seen.files, query.Table{Name: part.FormName(), Data: data}) + continue + } + seen.fields = append(seen.fields, [2]string{part.FormName(), string(data)}) + if part.FormName() == "query" { + seen.sql = string(data) + } + } + } f.mu.Lock() f.seen = append(f.seen, seen) f.mu.Unlock() @@ -100,20 +127,35 @@ func (f *fakeCH) sql() string { } // params are the most recent request's bound values in placeholder order: -// what {p0:…}, {p1:…}, … received. +// what {p0:…}, {p1:…}, … received, skipping a position bound as a table. func (f *fakeCH) params() []string { s := f.last() if s == nil { return nil } - var out []string - for i := 0; ; i++ { - v, ok := s.query["param_p"+strconv.Itoa(i)] - if !ok { - return out + var positions []int + for k := range s.query { + if n, ok := strings.CutPrefix(k, "param_p"); ok { + i, err := strconv.Atoi(n) + if err == nil { + positions = append(positions, i) + } } - out = append(out, v[0]) } + slices.Sort(positions) + var out []string + for _, i := range positions { + out = append(out, s.query.Get("param_p"+strconv.Itoa(i))) + } + return out +} + +// tables are the external tables of the most recent request. +func (f *fakeCH) tables() []query.Table { + if s := f.last(); s != nil { + return s.files + } + return nil } // setting is one query-string setting of the most recent request. @@ -201,20 +243,21 @@ func TestCHReader_Request(t *testing.T) { Headers: map[string]string{"X-Proxy-Token": "t", "X-ClickHouse-User": "spoofed"}, } _, err := ch.reader().do(t.Context(), target, 0, chRequest{ - sql: "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` IN {p1:Array(String)}", - params: []string{"/home", "['x','y']"}, + sql: "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` = {p1:String}", + params: []query.Param{{Name: "p0", Value: "/home"}, {Name: "p1", Value: `a\tb`}}, settings: map[string]string{"max_rows_to_read": "1", "read_overflow_mode": "throw"}, }) require.NoError(t, err) got := ch.last() - assert.Equal(t, "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` IN {p1:Array(String)}", got.sql) + assert.Equal(t, "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` = {p1:String}", got.sql) + assert.Equal(t, "text/plain; charset=utf-8", got.header.Get("Content-Type")) for name, want := range chReadSettingsFixed { assert.Equal(t, want, got.query.Get(name), name) } assert.Equal(t, "2", got.query.Get("readonly")) assert.Equal(t, "warehouse", got.query.Get("database")) - assert.Equal(t, []string{"/home", "['x','y']"}, ch.params()) + assert.Equal(t, []string{"/home", `a\tb`}, ch.params()) assert.Equal(t, "1", got.query.Get("max_rows_to_read")) assert.Equal(t, "throw", got.query.Get("read_overflow_mode")) assert.Equal(t, "u", got.header.Get("X-ClickHouse-User")) @@ -228,6 +271,41 @@ func TestCHReader_Request(t *testing.T) { assert.Equal(t, int32(1), ch.writes.Load()) } +// TestCHReader_ExternalTables pins the request that carries `in` lists: a +// multipart form whose query field is the statement, each table described by +// its _structure and _format fields before any file part — ClickHouse reads +// a part as it arrives — and then the tables' bytes untouched. The scalars +// and every setting stay on the query string. +func TestCHReader_ExternalTables(t *testing.T) { + t.Parallel() + ch := &fakeCH{} + sql := "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` IN (SELECT v FROM _p1) AND `c` IN (SELECT v FROM _p2)" + tables := []query.Table{{Name: "_p1", Data: []byte("\x01x\x02yz")}, {Name: "_p2", Data: []byte{}}} + _, err := ch.reader().do(t.Context(), chconn.Target{URL: fakeCHURL, Database: "warehouse"}, 0, chRequest{ + sql: sql, + params: []query.Param{{Name: "p0", Value: "/home"}}, + tables: tables, + }) + require.NoError(t, err) + + got := ch.last() + mt, _, err := mime.ParseMediaType(got.header.Get("Content-Type")) + require.NoError(t, err) + assert.Equal(t, "multipart/form-data", mt) + assert.Equal(t, [][2]string{ + {"query", sql}, + {"_p1_structure", query.TableStructure}, + {"_p1_format", query.TableFormat}, + {"_p2_structure", query.TableStructure}, + {"_p2_format", query.TableFormat}, + }, got.fields) + assert.Equal(t, tables, got.files) + assert.Equal(t, []string{"/home"}, ch.params()) + assert.Equal(t, "2", got.query.Get("readonly")) + assert.Equal(t, "warehouse", got.query.Get("database")) + assert.False(t, got.query.Has("query"), "the statement must not ride on the request line") +} + // TestCHReader_Errors covers every way a read fails, each typed so // chconn.Classify and chFailureOf can answer it by class, and each carrying // ClickHouse's own message — the only diagnostic an operator gets. @@ -375,27 +453,45 @@ func TestCHReader_ConnectionCapHolds(t *testing.T) { assert.Equal(t, 1, peak) } -// TestCheckParamSizes: a bound value is refused when its percent-encoded form +// TestCheckRequestSize: a scalar is refused when its percent-encoded form // passes ClickHouse's per-field limit — measured to the byte, 131072 read and -// 132096 refused — and all of them together when they pass what the request -// line holds. -func TestCheckParamSizes(t *testing.T) { +// 131073 refused — and all of them together when they pass what the request +// line holds; a statement with an `in` list when it passes the form field it +// travels in. An `in` list itself is never refused, whatever its size. +func TestCheckRequestSize(t *testing.T) { t.Parallel() - assert.NoError(t, checkParamSizes(nil)) - assert.NoError(t, checkParamSizes([]string{strings.Repeat("a", chMaxFieldBytes)})) - err := checkParamSizes([]string{strings.Repeat("a", chMaxFieldBytes+1)}) + scalars := func(vals ...string) *query.Bound { + b := &query.Bound{SQL: "SELECT 1"} + for i, v := range vals { + b.Params = append(b.Params, query.Param{Name: "p" + strconv.Itoa(i), Value: v}) + } + return b + } + assert.NoError(t, checkRequestSize(&query.Bound{})) + assert.NoError(t, checkRequestSize(scalars(strings.Repeat("a", chMaxFieldBytes)))) + err := checkRequestSize(scalars(strings.Repeat("a", chMaxFieldBytes+1))) require.Error(t, err) - assert.Contains(t, err.Error(), "in list") + assert.Contains(t, err.Error(), "filter value too large") // The encoded size counts: a quote is three bytes on the wire. - require.Error(t, checkParamSizes([]string{strings.Repeat("'", chMaxFieldBytes/3+1)})) + require.Error(t, checkRequestSize(scalars(strings.Repeat("'", chMaxFieldBytes/3+1)))) under := strings.Repeat("a", chMaxFieldBytes) var many []string for range (chMaxURIBytes - chURIHeadroom) / chMaxFieldBytes { many = append(many, under) } - assert.NoError(t, checkParamSizes(many)) - require.Error(t, checkParamSizes(append(many, under))) + assert.NoError(t, checkRequestSize(scalars(many...))) + require.Error(t, checkRequestSize(scalars(append(many, under)...))) + + huge := &query.Bound{SQL: "SELECT 1 WHERE c IN (SELECT v FROM _p0)", Tables: []query.Table{{Name: "_p0", Data: make([]byte, 4<<20)}}} + assert.NoError(t, checkRequestSize(huge), "an in list has no size cap") + long := &query.Bound{SQL: strings.Repeat(" ", chMaxFieldBytes), Tables: huge.Tables} + assert.NoError(t, checkRequestSize(long)) + long.SQL += " " + err = checkRequestSize(long) + require.Error(t, err) + assert.Contains(t, err.Error(), "query too large") + assert.NoError(t, checkRequestSize(&query.Bound{SQL: long.SQL}), "without a table the statement is the body, not a field") } // chExceptionBody is ClickHouse's own refusal text for code. diff --git a/internal/api/structured_query.go b/internal/api/structured_query.go index e0d969da..df59dd26 100644 --- a/internal/api/structured_query.go +++ b/internal/api/structured_query.go @@ -171,13 +171,14 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) } // Bind the built SQL for ClickHouse's HTTP interface: positional `?` - // placeholders become {pN:String} / {pN:Array(String)} named parameters - // and each value becomes the text ClickHouse reads it back from. A value - // with no text form (a JSON null, an object), or one too large for the - // HTTP interface to take, is a malformed query, not a server fault. - chSQL, chParams, err := result.NamedParams() + // placeholders become {pN:String} query parameters, and `in` lists + // external tables, each value the text ClickHouse reads it back from. A + // value with no text form (a JSON null, an object), or a query too large + // for the HTTP interface to take, is a malformed query, not a server + // fault. + bound, err := result.Bind() if err == nil { - err = checkParamSizes(chParams) + err = checkRequestSize(bound) } if err != nil { writeJSONError(w, http.StatusBadRequest, err.Error()) @@ -188,7 +189,7 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) // the singleflight key too. Keyed on what reaches ClickHouse, so two // requests that differ only in a spelling the binding erases share an // entry and two that differ in the bytes sent never do. - cacheKey := queryCacheKey(store.Tenant(), chSQL, chParams) + cacheKey := queryCacheKey(store.Tenant(), bound.SQL, cacheValues(bound)) // TODO: impl scope scope := "" @@ -265,7 +266,7 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) start := time.Now() // ClickHouse's own JSON rendering of the rows, stored and served // verbatim — no per-row scan, no re-marshal. - data, err := h.ch.do(queryCtx, target, conns, chRequest{sql: chSQL, params: chParams, settings: chSettings}) + data, err := h.ch.do(queryCtx, target, conns, chRequest{sql: bound.SQL, params: bound.Params, tables: bound.Tables, settings: chSettings}) if err != nil { return nil, err } diff --git a/internal/api/structured_query_integration_test.go b/internal/api/structured_query_integration_test.go new file mode 100644 index 00000000..55c72862 --- /dev/null +++ b/internal/api/structured_query_integration_test.go @@ -0,0 +1,327 @@ +//go:build integration + +package api + +import ( + "context" + "encoding/json" + "fmt" + "net" + "net/http" + "os" + "regexp" + "slices" + "strconv" + "strings" + "testing" + "time" + + "github.com/moby/moby/api/types/container" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/testcontainers/testcontainers-go" + "github.com/testcontainers/testcontainers-go/wait" + + "github.com/Wave-RF/WaveHouse/internal/chconn" + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/query" +) + +// filterCHImage is the server the differential runs against; +// WAVEHOUSE_TEST_CLICKHOUSE_IMAGE names another line to compare. +const filterCHImage = "clickhouse/clickhouse-server:26.8.15.10" + +// startFilterClickHouse runs a ClickHouse whose server zone is Europe/Berlin +// — not UTC, and different from every zone a test column declares — so a +// value read in the wrong zone lands on the wrong row. +func startFilterClickHouse(t *testing.T) chconn.Target { + t.Helper() + ctx := context.Background() + image := filterCHImage + if v := os.Getenv("WAVEHOUSE_TEST_CLICKHOUSE_IMAGE"); v != "" { + image = v + } + ctr, err := testcontainers.GenericContainer(ctx, testcontainers.GenericContainerRequest{ + ContainerRequest: testcontainers.ContainerRequest{ + Image: image, + ExposedPorts: []string{"8123/tcp"}, + Env: map[string]string{"CLICKHOUSE_PASSWORD": "test"}, + Files: []testcontainers.ContainerFile{{ + Reader: strings.NewReader("Europe/Berlin"), + ContainerFilePath: "/etc/clickhouse-server/config.d/timezone.xml", + FileMode: 0o644, + }}, + // The image's VOLUME would otherwise leave an anonymous volume behind. + HostConfigModifier: func(hc *container.HostConfig) { + hc.Tmpfs = map[string]string{"/var/lib/clickhouse": ""} + }, + WaitingFor: wait.ForHTTP("/ping").WithPort("8123/tcp").WithStatusCodeMatcher(func(status int) bool { + return status == http.StatusOK + }).WithStartupTimeout(120 * time.Second), + }, + Started: true, + }) + testcontainers.CleanupContainer(t, ctr) + require.NoError(t, err) + host, err := ctr.Host(ctx) + require.NoError(t, err) + port, err := ctr.MappedPort(ctx, "8123") + require.NoError(t, err) + return chconn.Target{URL: "http://" + net.JoinHostPort(host, port.Port()), Username: "default", Password: "test", Database: "default"} +} + +// filterDB is one ClickHouse and the reader the read path uses on it. +type filterDB struct { + t *testing.T + target chconn.Target + ch *chReader +} + +func (db *filterDB) exec(sql string) { + db.t.Helper() + _, err := db.ch.do(db.t.Context(), db.target, 1, chRequest{sql: sql, write: true}) + require.NoError(db.t, err, sql) +} + +// schema is table's schema as ClickHouse reports it, the types discovery +// reads. +func (db *filterDB) schema(table string) *discovery.TableSchema { + db.t.Helper() + data, err := db.ch.do(db.t.Context(), db.target, 1, chRequest{ + sql: "SELECT name, type FROM system.columns WHERE database = currentDatabase() AND table = {t:String} ORDER BY position", + params: []query.Param{{Name: "t", Value: table}}, + }) + require.NoError(db.t, err) + var cols []discovery.Column + require.NoError(db.t, json.Unmarshal(data, &cols)) + return &discovery.TableSchema{Name: table, Columns: cols} +} + +// bound builds q the way the structured query handler does. +func (db *filterDB) bound(table string, q query.StructuredQuery) *query.Bound { + db.t.Helper() + res, err := query.Build(table, &q, db.schema(table), nil, 0, query.DefaultMaxRows) + require.NoError(db.t, err) + b, err := res.Bind() + require.NoError(db.t, err) + require.NoError(db.t, checkRequestSize(b)) + return b +} + +func (db *filterDB) run(b *query.Bound, sql string) ([]byte, error) { + return db.ch.do(db.t.Context(), db.target, 1, chRequest{sql: sql, params: b.Params, tables: b.Tables, settings: chReadSettings(chQueryLimits{ExecutionTime: 20 * time.Second})}) +} + +// ids runs q on table and returns the id of every row, sorted. +func (db *filterDB) ids(table string, filters []query.Filter, tr *query.TimeRange) ([]int, error) { + db.t.Helper() + b := db.bound(table, query.StructuredQuery{Columns: query.Columns{"id"}, Filters: filters, TimeRange: tr}) + data, err := db.run(b, b.SQL) + if err != nil { + return nil, err + } + var rows []struct { + ID int `json:"id"` + } + require.NoError(db.t, json.Unmarshal(data, &rows)) + out := []int{} + for _, r := range rows { + out = append(out, r.ID) + } + slices.Sort(out) + return out, nil +} + +var granulesRe = regexp.MustCompile(`Granules: (\d+)/(\d+)`) + +// granules runs EXPLAIN indexes = 1 over q and returns the primary key's +// selected and total granules. +func (db *filterDB) granules(table string, filters []query.Filter, tr *query.TimeRange) (int, int) { + db.t.Helper() + b := db.bound(table, query.StructuredQuery{Columns: query.Columns{"id"}, Filters: filters, TimeRange: tr}) + data, err := db.run(b, "EXPLAIN indexes = 1 "+b.SQL) + require.NoError(db.t, err) + var rows []struct { + Explain string `json:"explain"` + } + require.NoError(db.t, json.Unmarshal(data, &rows)) + for _, r := range rows { + if m := granulesRe.FindStringSubmatch(r.Explain); m != nil { + sel, _ := strconv.Atoi(m[1]) + total, _ := strconv.Atoi(m[2]) + return sel, total + } + } + db.t.Fatalf("no primary key granules in EXPLAIN of %s", b.SQL) + return 0, 0 +} + +func filterOn(col, op string, v any) []query.Filter { + return []query.Filter{{Column: col, Op: op, Value: v}} +} + +func idSpan(from, to int) []int { + var out []int + for i := from; i <= to; i++ { + out = append(out, i) + } + return out +} + +// TestIntegration_FilterValuesAgainstClickHouse is the differential for the +// read path's filter binding: real rows, real primary keys, and a server zone +// that is not UTC. Each time table holds id k at 2026-06-21T00:00:00Z + k×30 +// min (k = 0…47), so 04:00Z is id 8 and 05:00Z id 10; tn also holds id 1000 +// at 04:00:00.500Z. A value parsed in the wrong zone lands hours away. +func TestIntegration_FilterValuesAgainstClickHouse(t *testing.T) { + db := &filterDB{t: t, target: startFilterClickHouse(t), ch: newCHReader(readerHTTPClient)} + for _, stmt := range []string{ + "CREATE TABLE tk (id UInt32, ts DateTime('Asia/Tokyo')) ENGINE = MergeTree ORDER BY ts SETTINGS index_granularity = 1", + "INSERT INTO tk SELECT number, toDateTime('2026-06-21 00:00:00', 'UTC') + number * 1800 FROM numbers(48)", + "CREATE TABLE tn (id UInt32, ts DateTime64(3, 'America/New_York')) ENGINE = MergeTree ORDER BY ts SETTINGS index_granularity = 1", + "INSERT INTO tn SELECT number, toDateTime64('2026-06-21 00:00:00', 3, 'UTC') + toIntervalSecond(number * 1800) FROM numbers(48)", + "INSERT INTO tn SELECT 1000, toDateTime64('2026-06-21 04:00:00.500', 3, 'UTC')", + "OPTIMIZE TABLE tn FINAL", + "CREATE TABLE tp (id UInt32, ts DateTime) ENGINE = MergeTree ORDER BY ts SETTINGS index_granularity = 1", + "INSERT INTO tp SELECT number, toDateTime('2026-06-21 00:00:00', 'UTC') + number * 1800 FROM numbers(48)", + "CREATE TABLE td (id UInt32, d Date) ENGINE = MergeTree ORDER BY d SETTINGS index_granularity = 1", + "INSERT INTO td SELECT number, toDate('2026-06-15') + number FROM numbers(14)", + "CREATE TABLE tv (id UInt32, s String, dec Decimal(10, 2), u8 UInt8) ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 1", + `INSERT INTO tv VALUES (1, 'plain', 12.50, 1), (2, 'it''s', 3, 2), (3, 'a\\b', 0, 3), (4, 'a\tb', 0, 4), (5, 'a\nb', 0, 5), (6, '\\N', 0, 6), (7, '', 0, 7)`, + } { + db.exec(stmt) + } + + // The column's zone, as each time table spells 04:00Z without one. + local := map[string]string{"tk": "2026-06-21 13:00:00", "tn": "2026-06-21 00:00:00", "tp": "2026-06-21 06:00:00"} + for _, table := range []string{"tk", "tn", "tp"} { + t.Run(table, func(t *testing.T) { + db := &filterDB{t: t, target: db.target, ch: db.ch} + // tn's id 1000 sits at 04:00:00.500Z, between ids 8 and 9. + half := func(ids ...int) []int { + if table == "tn" { + ids = append(ids, 1000) + } + return ids + } + cases := []struct { + name string + filters []query.Filter + tr *query.TimeRange + want []int + }{ + {"eq Z", filterOn("ts", "eq", "2026-06-21T04:00:00Z"), nil, []int{8}}, + {"eq offset", filterOn("ts", "eq", "2026-06-21T13:00:00+09:00"), nil, []int{8}}, + {"eq zone-less, read in the column's zone", filterOn("ts", "eq", local[table]), nil, []int{8}}, + {"eq unix seconds", filterOn("ts", "eq", json.Number("1782014400")), nil, []int{8}}, + {"gt a fraction", filterOn("ts", "gt", "2026-06-21T04:00:00.5Z"), nil, idSpan(9, 47)}, + {"gte a fraction", filterOn("ts", "gte", "2026-06-21T04:00:00.5Z"), nil, half(idSpan(9, 47)...)}, + {"lt a fraction", filterOn("ts", "lt", "2026-06-21T04:00:00.5Z"), nil, idSpan(0, 8)}, + {"eq a fraction", filterOn("ts", "eq", "2026-06-21T04:00:00.5Z"), nil, half()}, + {"in", filterOn("ts", "in", []any{"2026-06-21T04:00:00Z", "2026-06-21T14:00:00+09:00"}), nil, []int{8, 10}}, + {"in a fraction", filterOn("ts", "in", []any{"2026-06-21T04:00:00.5Z"}), nil, half()}, + {"time_range", nil, &query.TimeRange{Column: "ts", Since: "2026-06-21T04:00:00Z", Until: "2026-06-21T05:00:00Z"}, half(8, 9, 10)}, + } + for _, c := range cases { + got, err := db.ids(table, c.filters, c.tr) + require.NoError(t, err, c.name) + want := c.want + slices.Sort(want) + if want == nil { + want = []int{} + } + assert.Equal(t, want, got, c.name) + } + + // The primary key narrows every shape the builder emits. + for _, c := range []struct { + name string + filters []query.Filter + tr *query.TimeRange + }{ + {"eq", filterOn("ts", "eq", "2026-06-21T04:00:00Z"), nil}, + {"time_range", nil, &query.TimeRange{Column: "ts", Since: "2026-06-21T04:00:00Z", Until: "2026-06-21T05:00:00Z"}}, + {"in", filterOn("ts", "in", []any{"2026-06-21T04:00:00Z", "2026-06-21T14:00:00+09:00"}), nil}, + } { + sel, total := db.granules(table, c.filters, c.tr) + assert.Less(t, sel, total/4, "%s: the primary key must narrow the read (%d/%d granules)", c.name, sel, total) + } + + // A value ClickHouse cannot parse is its refusal, classed as one. + for _, filters := range [][]query.Filter{filterOn("ts", "eq", "banana"), filterOn("ts", "in", []any{"2026-06-21T04:00:00Z", "banana"})} { + _, err := db.ids(table, filters, nil) + require.Error(t, err) + assert.Equal(t, chconn.Rejected, chconn.Classify(err), "%v", err) + } + + // An in list far past the 128 KiB a query parameter takes. + big := make([]any, 0, 60048) + for k := range 48 { + big = append(big, time.Date(2026, 6, 21, 0, 0, 0, 0, time.UTC).Add(time.Duration(k)*30*time.Minute).Format(time.RFC3339)) + } + for k := range 60000 { + big = append(big, time.Date(2027, 1, 1, 0, 0, 0, 0, time.UTC).Add(time.Duration(k)*time.Second).Format(time.RFC3339)) + } + got, err := db.ids(table, filterOn("ts", "in", big), nil) + require.NoError(t, err) + assert.Equal(t, idSpan(0, 47), got) + }) + } + + t.Run("Date", func(t *testing.T) { + db := &filterDB{t: t, target: db.target, ch: db.ch} + for _, c := range []struct { + name string + v any + want []int + }{ + {"a date", "2026-06-21", []int{6}}, + {"an instant, on its date in the server's zone", "2026-06-21T04:00:00Z", []int{6}}, + } { + got, err := db.ids("td", filterOn("d", "eq", c.v), nil) + require.NoError(t, err, c.name) + assert.Equal(t, c.want, got, c.name) + } + got, err := db.ids("td", filterOn("d", "in", []any{"2026-06-21", "2026-06-23T10:00:00Z"}), nil) + require.NoError(t, err) + assert.Equal(t, []int{6, 8}, got) + sel, total := db.granules("td", filterOn("d", "in", []any{"2026-06-21", "2026-06-23T10:00:00Z"}), nil) + assert.Less(t, sel, total/2, "%d/%d granules", sel, total) + }) + + t.Run("values arrive as written", func(t *testing.T) { + db := &filterDB{t: t, target: db.target, ch: db.ch} + values := map[int]string{1: "plain", 2: "it's", 3: `a\b`, 4: "a\tb", 5: "a\nb", 6: `\N`, 7: ""} + var all []any + for id, v := range values { + got, err := db.ids("tv", filterOn("s", "eq", v), nil) + require.NoError(t, err, "%q", v) + assert.Equal(t, []int{id}, got, "eq %q", v) + all = append(all, v) + } + got, err := db.ids("tv", filterOn("s", "in", all), nil) + require.NoError(t, err) + assert.Equal(t, idSpan(1, 7), got) + + // An in list compares under the column's type, not as text: 3.50 is + // the Decimal 3.00's neighbour, 3 and 12.5 are its values; 256 is no + // UInt8 and matches nothing. + got, err = db.ids("tv", filterOn("dec", "in", []any{json.Number("12.5"), "3.00", "3.50"}), nil) + require.NoError(t, err) + assert.Equal(t, []int{1, 2}, got) + got, err = db.ids("tv", filterOn("u8", "in", []any{json.Number("2"), json.Number("256")}), nil) + require.NoError(t, err) + assert.Equal(t, []int{2}, got) + sel, total := db.granules("tv", filterOn("id", "in", []any{json.Number("2"), json.Number("5")}), nil) + assert.Less(t, sel, total, "%d/%d granules", sel, total) + + big := make([]any, 0, 100000) + for i := range 100000 { + big = append(big, fmt.Sprintf("value-%d", i)) + } + big = append(big, "a\tb") + got, err = db.ids("tv", filterOn("s", "in", big), nil) + require.NoError(t, err) + assert.Equal(t, []int{4}, got) + }) +} diff --git a/internal/api/structured_query_test.go b/internal/api/structured_query_test.go index f3c3774c..6f8d4cc1 100644 --- a/internal/api/structured_query_test.go +++ b/internal/api/structured_query_test.go @@ -623,26 +623,24 @@ func TestStructuredQuery_FilterValuesKeepTheirDigits(t *testing.T) { w := httptest.NewRecorder() h.Handle(w, withTenant(r)) require.Equal(t, http.StatusOK, w.Code, w.Body.String()) - assert.Equal(t, []string{"12.50", "['9007199254740993']"}, ch.params()) + assert.Equal(t, []string{"12.50"}, ch.params()) + assert.Equal(t, []query.Table{{Name: "_p1", Data: append([]byte{16}, "9007199254740993"...)}}, ch.tables()) } // TestStructuredQuery_UnbindableFilterValueIs400: a filter value with no -// honest binding — a JSON null, or an `in` list too large for ClickHouse's -// HTTP interface to take — is the caller's malformed query, a 400 before -// anything reaches ClickHouse, rather than a server error after. +// honest binding — a JSON null, or a scalar too large for ClickHouse's HTTP +// interface to take — is the caller's malformed query, a 400 before anything +// reaches ClickHouse, rather than a server error after. func TestStructuredQuery_UnbindableFilterValueIs400(t *testing.T) { t.Parallel() - huge := make([]any, 0, 20000) - for i := range 20000 { - huge = append(huge, fmt.Sprintf("v%d", i)) - } for _, tc := range []struct { name string filter query.Filter want string }{ {"null value", query.Filter{Column: "page", Op: "eq", Value: nil}, "must not be null"}, - {"oversized in list", query.Filter{Column: "page", Op: "in", Value: huge}, "split a long in list"}, + {"null in an in list", query.Filter{Column: "page", Op: "in", Value: []any{"a", nil}}, "must not be null"}, + {"oversized scalar", query.Filter{Column: "page", Op: "eq", Value: strings.Repeat("a", chMaxFieldBytes+1)}, "filter value too large"}, } { t.Run(tc.name, func(t *testing.T) { t.Parallel() @@ -657,3 +655,46 @@ func TestStructuredQuery_UnbindableFilterValueIs400(t *testing.T) { }) } } + +// TestStructuredQuery_LargeInListReachesClickHouse: an `in` list far past +// ClickHouse's 128 KiB parameter limit — the realistic victim of that limit — +// reaches ClickHouse whole, as an external table, alongside the request's +// scalars on the query string. Only the 1 MiB request body bounds it. +func TestStructuredQuery_LargeInListReachesClickHouse(t *testing.T) { + t.Parallel() + const n = 60000 + huge := make([]any, 0, n) + for i := range n { + huge = append(huge, fmt.Sprintf("v%d", i)) + } + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{Columns: []string{"page"}, Filters: []query.Filter{ + {Column: "user_id", Op: "eq", Value: "u1"}, + {Column: "page", Op: "in", Value: huge}, + }}))) + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + assert.Equal(t, "SELECT `page` FROM `clicks` WHERE `user_id` = {p0:String} AND `page` IN (SELECT v FROM _p1) LIMIT 10000", ch.sql()) + assert.Equal(t, []string{"u1"}, ch.params()) + tables := ch.tables() + require.Len(t, tables, 1) + assert.Equal(t, "_p1", tables[0].Name) + assert.Greater(t, len(tables[0].Data), chMaxFieldBytes*3, "the list must not be capped at a query parameter's size") +} + +// TestStructuredQuery_TimestampFilterParsesInClickHouse: a filter value on a +// DateTime column reaches ClickHouse as the caller wrote it, under the parse +// that reads it in the column's zone. +func TestStructuredQuery_TimestampFilterParsesInClickHouse(t *testing.T) { + t.Parallel() + ch := &fakeCH{} + h := newCapturingHandler(t, ch, policyWithViewer(policy.SelectPermissions{AllowColumns: []string{"*"}})) + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRequest(t, query.StructuredQuery{Columns: []string{"page"}, Filters: []query.Filter{ + {Column: "ts", Op: "gte", Value: "2026-06-21T04:00:00.5+09:00"}, + }}))) + require.Equal(t, http.StatusOK, w.Code, w.Body.String()) + assert.Equal(t, "SELECT `page` FROM `clicks` WHERE `ts` >= parseDateTime64BestEffort({p0:String}, 8) LIMIT 10000", ch.sql()) + assert.Equal(t, []string{"2026-06-21T04:00:00.5+09:00"}, ch.params()) +} diff --git a/internal/query/bind.go b/internal/query/bind.go new file mode 100644 index 00000000..906b2fbb --- /dev/null +++ b/internal/query/bind.go @@ -0,0 +1,374 @@ +package query + +import ( + "encoding/binary" + "encoding/json" + "fmt" + "strconv" + "strings" + + "github.com/Wave-RF/WaveHouse/internal/chsql" +) + +// ─── ClickHouse binding ────────────────────────────────────────────────────── + +// Bound is a built query in the form ClickHouse's HTTP interface takes: the +// SQL with named placeholders, the query parameters they read, and the +// external tables its `in` lists read. +type Bound struct { + SQL string + // Params supply param_, already encoded for ClickHouse's parameter + // reader, in placeholder order. + Params []Param + // Tables are the `in` lists, in placeholder order. + Tables []Table +} + +// Param is one query parameter: the value of param_. +type Param struct { + Name, Value string +} + +// Table is one `in` list sent as a ClickHouse external table called Name: +// TableStructure, one row per element, in TableFormat — each element's bytes +// as written, length-prefixed, so nothing in a value needs escaping. +type Table struct { + Name string + Data []byte +} + +// The shape of every Table, which the SQL reads as `SELECT … v FROM `. +const ( + TableStructure = "v String" + TableFormat = "RowBinary" +) + +// listParam is an `in` list: one external table, its elements converted to +// the column's type in the subquery that reads it. +type listParam struct { + Values []any + Conv conversion +} + +// convertedParam is a scalar bound as {pN:String} whose placeholder expands +// to Conv's expression around it. +type convertedParam struct { + Value any + Conv conversion +} + +// conversionKind is how ClickHouse turns a bound String into a column's type. +type conversionKind int + +const ( + // asString leaves it to ClickHouse's comparison, which converts a String + // operand to the column's type. An `in` list on such a column is a set of + // Strings, which only a String column compares as typed. + asString conversionKind = iota + // parseTime parses with parseDateTime64BestEffort: an offset or Z gives + // the exact instant, and a zone-less value reads in the column's declared + // zone, else the server's — as it does compared to the column directly. + parseTime + // parseDate and parseDate32 take the calendar date of parseTime's + // instant, in the server's zone. + parseDate + parseDate32 + // castTo converts each element of an `in` list to the column's type with + // accurateCastOrNull. A subquery's set keeps its own type and ClickHouse + // casts the column to it, so a set of Strings would compare text: + // Decimal 12.50 against '12.50' missed (measured on 26.8.15.10). CAST + // instead wrapped '256' to 0 on a UInt8 column; accurateCastOrNull turns + // it into NULL, which matches nothing, as the literal list did. + castTo +) + +// conversion is how ClickHouse turns a filter's bound String into the +// filtered column's type, read off the column's type in the schema. +type conversion struct { + kind conversionKind + scale int // parseTime: the parse's scale + zone string // parseTime: the column's declared zone, "" for the server's + typ string // castTo: the column's type without Nullable/LowCardinality +} + +// The scales a parseTime conversion parses at. Fine enough to compare any +// value exactly at the column's precision; scale 9 is used only for a +// DateTime64(9) column, because it tops out at 2262-04-11 and a +// DateTime64(3) value past that failed the whole comparison with +// DECIMAL_OVERFLOW (measured on 24.8.14.39 and 26.8.15.10), where scale 8 +// holds DateTime64's whole range. Scale 0 is no use: it parses into a +// DateTime, so `>= …04:00:00.5Z` admitted 04:00:00, and it floors at 1970. +const ( + timeScale = 8 + timeScaleNano = 9 +) + +// conversionFor picks the conversion for a column of type colType, as +// system.columns spells it. A type it cannot read keeps asString, which +// leaves ClickHouse to refuse what it cannot compare rather than guess a +// zone. +func conversionFor(colType string) conversion { + t := colType + for { + inner, ok := unwrapType(t, "Nullable(") + if !ok { + inner, ok = unwrapType(t, "LowCardinality(") + } + if !ok { + break + } + t = inner + } + switch { + case t == "Date": + return conversion{kind: parseDate} + case t == "Date32": + return conversion{kind: parseDate32} + case t == "DateTime": + return conversion{kind: parseTime, scale: timeScale} + case strings.HasPrefix(t, "DateTime(") || strings.HasPrefix(t, "DateTime64("): + if c, ok := dateTimeConversion(t); ok { + return c + } + return conversion{kind: asString} + case t == "String" || t == "": + return conversion{kind: asString} + default: + return conversion{kind: castTo, typ: t} + } +} + +// dateTimeConversion reads `DateTime('zone')`, `DateTime64(p)` and +// `DateTime64(p, 'zone')`. +func dateTimeConversion(t string) (conversion, bool) { + open := strings.IndexByte(t, '(') + if !strings.HasSuffix(t, ")") { + return conversion{}, false + } + args := strings.Split(t[open+1:len(t)-1], ",") + c := conversion{kind: parseTime, scale: timeScale} + zoneArg := "" + if strings.HasPrefix(t, "DateTime64(") { + if len(args) > 2 { + return conversion{}, false + } + p, err := strconv.Atoi(strings.TrimSpace(args[0])) + if err != nil { + return conversion{}, false + } + if p == 9 { + c.scale = timeScaleNano + } + if len(args) == 2 { + zoneArg = args[1] + } + } else { + if len(args) != 1 { + return conversion{}, false + } + zoneArg = args[0] + } + if zoneArg != "" { + z := strings.TrimSpace(zoneArg) + if len(z) < 2 || z[0] != '\'' || z[len(z)-1] != '\'' || strings.ContainsAny(z[1:len(z)-1], `'\`) { + return conversion{}, false + } + c.zone = z[1 : len(z)-1] + } + return c, true +} + +func unwrapType(t, prefix string) (string, bool) { + if strings.HasPrefix(t, prefix) && strings.HasSuffix(t, ")") { + return t[len(prefix) : len(t)-1], true + } + return "", false +} + +// scalar is v as Build binds it for this column: wrapped when ClickHouse must +// parse it, as written otherwise. +func (c conversion) scalar(v any) any { + switch c.kind { + case parseTime, parseDate, parseDate32: + return convertedParam{Value: v, Conv: c} + case asString, castTo: + } + return v +} + +// expr is the SQL converting arg, a String, to the column's type. For +// asString and castTo it is a scalar's own placeholder — ClickHouse's +// comparison converts it — and only an `in` element is cast (element). +func (c conversion) expr(arg string) string { + switch c.kind { + case parseTime: + if c.zone == "" { + return fmt.Sprintf("parseDateTime64BestEffort(%s, %d)", arg, c.scale) + } + return fmt.Sprintf("parseDateTime64BestEffort(%s, %d, %s)", arg, c.scale, sqlString(c.zone)) + case parseDate: + return fmt.Sprintf("toDate(parseDateTime64BestEffort(%s, %d))", arg, timeScale) + case parseDate32: + return fmt.Sprintf("toDate32(parseDateTime64BestEffort(%s, %d))", arg, timeScale) + case asString, castTo: + } + return arg +} + +// element is the SQL converting one `in` element, the String arg, to the +// column's type. +func (c conversion) element(arg string) string { + if c.kind == castTo { + return "accurateCastOrNull(" + arg + ", " + sqlString(c.typ) + ")" + } + return c.expr(arg) +} + +// sqlString renders s as a ClickHouse string literal. Only a schema's own +// type and zone names reach it, never a caller's value. +func sqlString(s string) string { + return "'" + strings.NewReplacer(`\`, `\\`, `'`, `\'`).Replace(s) + "'" +} + +// Bind rewrites the positional `?` placeholders in the built SQL into the +// form ClickHouse's HTTP interface takes, and renders each bound value as the +// text ClickHouse reads it back from. Placeholder i is named pN with N = i: +// +// - a scalar binds as {pN:String}. String is not a weaker binding than the +// column's own type: ClickHouse converts the parameter to the column's +// type for the comparison, so `UInt8 = {p:String}` with "256" is false +// and with "1.5" is a type error, matching what the server answers for +// the same literal. One rule, shared with the row filters the type layer +// compiles. +// - a value on a Date or DateTime column binds the same way and expands to +// the parse its conversion names, so ClickHouse reads the caller's own +// spelling — an RFC 3339 instant, or a zone-less time in the column's +// zone — instead of Go rewriting it. Compared directly, ClickHouse +// refuses RFC 3339 on DateTime and DateTime64 (TYPE_MISMATCH on +// 24.8.14.39 and 26.8.15.10). +// - a policy claim on an integer column (chsql.IntParam) expands to +// chsql.StrictInt over its one {pN:String}, because the plain form wraps +// a value at or past 2^64. +// - an `in` list becomes the external table _pN, read by +// `(SELECT FROM _pN)`. A table has no size cap, where a query +// parameter is capped at 128 KiB, and a subquery's set uses the primary +// key, where `c IN arrayMap(…)` read every granule on 26.8.15.10 and was +// refused on 24.8.14.39. +// +// The rewrite is a left-to-right scan for `?`, which is exact for this SQL and +// only for this SQL: Build never renders a value or a string literal, and +// chsql.BindUnsafe rejects a `?` in any identifier it quotes. The literals a +// conversion renders are written after the scan has passed them. +func (r *BuildResult) Bind() (*Bound, error) { + out := &Bound{} + if len(r.Params) == 0 { + if strings.Contains(r.SQL, "?") { + return nil, fmt.Errorf("query has placeholders but no bound values") + } + out.SQL = r.SQL + return out, nil + } + + var b strings.Builder + b.Grow(len(r.SQL) + len(r.Params)*16) + rest := r.SQL + for i, v := range r.Params { + q := strings.IndexByte(rest, '?') + if q < 0 { + return nil, fmt.Errorf("query has %d bound values but only %d placeholders", len(r.Params), i) + } + placeholder, err := out.bind("p"+strconv.Itoa(i), v) + if err != nil { + return nil, err + } + b.WriteString(rest[:q]) + b.WriteString(placeholder) + rest = rest[q+1:] + } + if strings.Contains(rest, "?") { + return nil, fmt.Errorf("query has more placeholders than the %d bound values", len(r.Params)) + } + b.WriteString(rest) + out.SQL = b.String() + return out, nil +} + +// bind adds v, bound under name, to out, and returns the SQL its placeholder +// becomes. +func (out *Bound) bind(name string, v any) (string, error) { + switch val := v.(type) { + case listParam: + data, err := rowBinaryStrings(val.Values) + if err != nil { + return "", err + } + table := "_" + name + out.Tables = append(out.Tables, Table{Name: table, Data: data}) + return "(SELECT " + val.Conv.element("v") + " FROM " + table + ")", nil + case chsql.IntParam: + out.Params = append(out.Params, Param{Name: name, Value: chsql.EscapeStringParam(val.Value)}) + return chsql.StrictInt(name, val.Type), nil + case convertedParam: + raw, err := chScalarText(val.Value) + if err != nil { + return "", err + } + out.Params = append(out.Params, Param{Name: name, Value: chsql.EscapeStringParam(raw)}) + return val.Conv.expr("{" + name + ":String}"), nil + } + raw, err := chScalarText(v) + if err != nil { + return "", err + } + out.Params = append(out.Params, Param{Name: name, Value: chsql.EscapeStringParam(raw)}) + return "{" + name + ":String}", nil +} + +// chScalarText is one scalar's value as plain text, before any encoding — +// what the caller means, not what the wire needs. +func chScalarText(v any) (string, error) { + switch val := v.(type) { + case string: + return val, nil + case json.Number: + // The caller's own digits, not a float64 round-trip: 12.50 stays + // "12.50" and an integer past 2^53 keeps every digit. + return val.String(), nil + case bool: + return strconv.FormatBool(val), nil + case float64: + return strconv.FormatFloat(val, 'f', -1, 64), nil + case int: + return strconv.Itoa(val), nil + case int64: + return strconv.FormatInt(val, 10), nil + case uint64: + return strconv.FormatUint(val, 10), nil + case nil: + // `col = NULL` is never true in SQL, so a null used to answer "no + // rows"; an empty String parameter would instead compare against the + // empty string, which is a different question. Refuse it (→ 400) + // rather than answer a question the caller did not ask. + return "", fmt.Errorf("filter value must not be null") + default: + return "", fmt.Errorf("unsupported filter value type %T", v) + } +} + +// rowBinaryStrings renders a list as RowBinary rows of one String column: each +// element's length as a varint, then its bytes. +func rowBinaryStrings(vals []any) ([]byte, error) { + var b []byte + for _, v := range vals { + if _, isList := v.([]any); isList { + return nil, fmt.Errorf("nested list in an 'in' value") + } + text, err := chScalarText(v) + if err != nil { + return nil, err + } + b = binary.AppendUvarint(b, uint64(len(text))) + b = append(b, text...) + } + return b, nil +} diff --git a/internal/query/builder.go b/internal/query/builder.go index e1aecab8..e2107ab5 100644 --- a/internal/query/builder.go +++ b/internal/query/builder.go @@ -1,7 +1,6 @@ package query import ( - "encoding/json" "fmt" "regexp" "strconv" @@ -23,8 +22,8 @@ const DefaultMaxRows = 10000 // BuildResult holds the generated SQL and its bound values. The SQL carries // positional `?` placeholders, one per entry in Params, in left-to-right -// order. NamedParams turns that pair into the named-parameter form -// ClickHouse's HTTP interface takes. +// order. Bind turns that pair into the form ClickHouse's HTTP interface +// takes. type BuildResult struct { SQL string Params []any @@ -47,7 +46,8 @@ type BuildResult struct { // Every identifier that reaches the SQL — columns, the table, aggregation // aliases — is backtick-quoted via chsql.QuoteIdent, so the builder accepts any // name ClickHouse accepts while remaining injection-safe. Values stay positional -// `?` parameters, which NamedParams turns into ClickHouse named parameters. +// `?` parameters, which Bind turns into ClickHouse query parameters and +// external tables. // // Projection rules: SelectAll requests every readable column (expanded to the // role's allow/deny set); an explicit Columns list projects exactly those (where @@ -240,7 +240,7 @@ func validateAndAuthorizeColumns(q *StructuredQuery, colSet map[string]bool, per } // The alias is backtick-quoted by aggregationExpr, so any legal ClickHouse // name is safe against injection. The lone refusal is a '?', which would - // shift the positional-to-named parameter rewrite (see NamedParams). + // shift the positional-to-named parameter rewrite (see Bind). if chsql.BindUnsafe(a.Alias) { return fmt.Errorf("unsupported aggregation alias (contains '?'): %s", a.Alias) } @@ -316,9 +316,10 @@ func resolveProjection(q *StructuredQuery, schema *discovery.TableSchema, perms func buildWhere(filters []Filter, timeRange *TimeRange, schema *discovery.TableSchema, bucketSeconds int) ([]string, []any, error) { var parts []string var params []any + typeOf := columnType(schema) for _, f := range filters { - clause, p, err := filterToSQL(f, isDateTimeColumn(schema, f.Column)) + clause, p, err := filterToSQL(f, conversionFor(typeOf(f.Column))) if err != nil { return nil, nil, fmt.Errorf("filter on column %q: %w", f.Column, err) } @@ -332,12 +333,13 @@ func buildWhere(filters []Filter, timeRange *TimeRange, schema *discovery.TableS if timeRange != nil && timeRange.Column != "" && timeRange.Since != "" { col := chsql.QuoteIdent(timeRange.Column) + conv := conversionFor(typeOf(timeRange.Column)) sinceTime, err := resolveTimeValue(timeRange.Since, bucketSeconds) if err != nil { return nil, nil, fmt.Errorf("time_range since: %w", err) } parts = append(parts, fmt.Sprintf("%s >= ?", col)) - params = append(params, sinceTime) + params = append(params, conv.scalar(sinceTime)) if timeRange.Until != "" { untilTime, err := resolveTimeValue(timeRange.Until, bucketSeconds) @@ -345,47 +347,37 @@ func buildWhere(filters []Filter, timeRange *TimeRange, schema *discovery.TableS return nil, nil, fmt.Errorf("time_range until: %w", err) } parts = append(parts, fmt.Sprintf("%s <= ?", col)) - params = append(params, untilTime) + params = append(params, conv.scalar(untilTime)) } } return parts, params, nil } -func filterToSQL(f Filter, dateTime bool) (string, []any, error) { +// filterToSQL renders one caller filter. conv is how ClickHouse turns the +// bound String into the column's type: a Date or DateTime column parses it +// explicitly, and an `in` list is read from an external table (see Bind). +// like compares text, so its value is never converted. +func filterToSQL(f Filter, conv conversion) (string, []any, error) { col := chsql.QuoteIdent(f.Column) - val := f.Value - if dateTime { - val = coerceFilterValue(val) - } switch strings.ToLower(f.Op) { case "eq": - return col + " = ?", []any{val}, nil + return col + " = ?", []any{conv.scalar(f.Value)}, nil case "neq": - return col + " != ?", []any{val}, nil + return col + " != ?", []any{conv.scalar(f.Value)}, nil case "gt": - return col + " > ?", []any{val}, nil + return col + " > ?", []any{conv.scalar(f.Value)}, nil case "gte": - return col + " >= ?", []any{val}, nil + return col + " >= ?", []any{conv.scalar(f.Value)}, nil case "lt": - return col + " < ?", []any{val}, nil + return col + " < ?", []any{conv.scalar(f.Value)}, nil case "lte": - return col + " <= ?", []any{val}, nil + return col + " <= ?", []any{conv.scalar(f.Value)}, nil case "like": - return col + " LIKE ?", []any{val}, nil + return col + " LIKE ?", []any{f.Value}, nil case "in": - // One placeholder for the whole list, bound as one Array(String) - // parameter rather than a parameter per element, so a long list - // costs one query-string field. if vals, ok := f.Value.([]any); ok && len(vals) > 0 { - if dateTime { - coerced := make([]any, len(vals)) - for i, v := range vals { - coerced[i] = coerceFilterValue(v) - } - vals = coerced - } - return col + " IN ?", []any{vals}, nil + return col + " IN ?", []any{listParam{Values: vals, Conv: conv}}, nil } return "", nil, fmt.Errorf("invalid value for 'in' operator") default: @@ -393,38 +385,6 @@ func filterToSQL(f Filter, dateTime bool) (string, []any, error) { } } -// coerceFilterValue rewrites a filter value on a DateTime or DateTime64 column -// that is an RFC3339 timestamp into ClickHouse's own DateTime spelling, in -// UTC, keeping its sub-second digits. ClickHouse reads a String compared -// against such a column with its basic parser, which refuses the RFC3339 -// spelling: measured on 24.8.14.39 and 26.6.3.62, a `T`, a `Z` or an offset -// is TYPE_MISMATCH on DateTime, and on DateTime64 too on 24.8 and inside an -// `in` list on both. The rewritten spelling parses on every one of them. -// -// A value that isn't a timestamp (a plain string, a number, etc.) is a valid -// non-temporal filter value, so the parse "failure" is just the expected -// non-timestamp case — pass it through unchanged rather than treat it as an error. -func coerceFilterValue(v any) any { - s, ok := v.(string) - if !ok { - return v - } - // RFC3339Nano parses both fractional and whole-second RFC3339 input. - if t, err := time.Parse(time.RFC3339Nano, s); err == nil { - return formatClickHouseTime(t) - } - return v -} - -// clickHouseDateTimeLayout renders a time in ClickHouse's native DateTime text -// format. The fractional ".999999999" preserves sub-second precision when -// present and drops trailing zeros, so a whole-second time has no decimal point. -const clickHouseDateTimeLayout = "2006-01-02 15:04:05.999999999" - -func formatClickHouseTime(t time.Time) string { - return t.UTC().Format(clickHouseDateTimeLayout) -} - // dayWeekRe matches a duration component with a day ("d") or week ("w") unit — // the two units time.ParseDuration rejects (it stops at hours). The magnitude // may be fractional, e.g. "0.5d". @@ -453,27 +413,25 @@ func expandDayWeek(s string) string { } // resolveTimeValue parses an RFC3339 timestamp or a relative duration like "1h", -// "30m", "7d" or "2w" and renders it as a ClickHouse DateTime literal (see -// formatClickHouseTime). When bucketSeconds > 0, timestamps are bucketed -// (truncated) to the nearest boundary. +// "30m", "7d" or "2w" and renders the instant as RFC 3339 in UTC, which the +// column's conversion parses exactly whatever the column's zone. When +// bucketSeconds > 0, timestamps are bucketed (truncated) to the nearest +// boundary. // // A value that is neither a duration nor a timestamp is rejected with an error // (which the builder surfaces as a 400) rather than returned unchanged: passing // a raw string to ClickHouse surfaces as an opaque DateTime parse error (#285). -// -// The output deliberately matches coerceFilterValue's format rather than -// RFC3339, which ClickHouse refuses on a DateTime column (see there). func resolveTimeValue(val string, bucketSeconds int) (string, error) { // Try a relative duration first (e.g., "1h", "30m", "7d", "2w"). Go's // time.ParseDuration only understands units up to hours, so day/week // suffixes are pre-expanded to hours. if d, err := time.ParseDuration(expandDayWeek(val)); err == nil { - return formatClickHouseTime(bucketTime(time.Now().UTC().Add(-d), bucketSeconds)), nil + return bucketTime(time.Now().UTC().Add(-d), bucketSeconds).Format(time.RFC3339Nano), nil } // Try an absolute timestamp (RFC3339Nano accepts fractional and whole-second // input); normalise to UTC before bucketing. if t, err := time.Parse(time.RFC3339Nano, val); err == nil { - return formatClickHouseTime(bucketTime(t.UTC(), bucketSeconds)), nil + return bucketTime(t.UTC(), bucketSeconds).Format(time.RFC3339Nano), nil } return "", fmt.Errorf("invalid time value %q: want an RFC3339 timestamp or a relative duration such as \"1h\", \"30m\", \"7d\", \"2w\"", val) } @@ -496,27 +454,6 @@ func columnType(schema *discovery.TableSchema) func(string) string { } } -// isDateTimeColumn reports whether the schema types column as DateTime or -// DateTime64, possibly Nullable or LowCardinality. -func isDateTimeColumn(schema *discovery.TableSchema, column string) bool { - c, ok := schema.Lookup(column) - if !ok { - return false - } - t := c.Type - for { - switch { - case strings.HasPrefix(t, "Nullable(") && strings.HasSuffix(t, ")"): - t = t[len("Nullable(") : len(t)-1] - continue - case strings.HasPrefix(t, "LowCardinality(") && strings.HasSuffix(t, ")"): - t = t[len("LowCardinality(") : len(t)-1] - continue - } - return strings.HasPrefix(t, "DateTime") - } -} - func schemaColumnSet(schema *discovery.TableSchema) map[string]bool { m := make(map[string]bool, len(schema.Columns)) for _, c := range schema.Columns { @@ -563,151 +500,3 @@ func isValidAggFn(fn string) bool { } return false } - -// ─── ClickHouse named-parameter binding ───────────────────────────────────── - -// NamedParams rewrites the positional `?` placeholders in the built SQL into -// ClickHouse named parameters and renders each bound value as the text -// ClickHouse will read it back from. It returns the rewritten SQL and the -// values for `param_p0` … `param_pN-1`, positionally. -// -// Every scalar binds as `{pN:String}` and every list as `{pN:Array(String)}` -// — one rule, shared with the row filters the type layer compiles. String is -// not a weaker binding than the column's own type: ClickHouse converts the -// parameter to the column's type for the comparison, so `UInt8 = {p:String}` -// with "256" is false and with "1.5" is a type error, matching what the -// server answers for the same literal. The exception is a policy claim on an -// integer column (a chsql.IntParam): its placeholder expands to -// chsql.StrictInt over the one `{pN:String}` parameter, because the plain -// form wraps a value at or past 2^64. -// -// The rewrite is a left-to-right scan for `?`, which is exact for this SQL and -// only for this SQL: Build never renders a value or a string literal, and -// chsql.BindUnsafe rejects a `?` in any identifier it quotes. -func (r *BuildResult) NamedParams() (string, []string, error) { - if len(r.Params) == 0 { - if strings.Contains(r.SQL, "?") { - return "", nil, fmt.Errorf("query has placeholders but no bound values") - } - return r.SQL, nil, nil - } - - params := make([]string, 0, len(r.Params)) - var b strings.Builder - b.Grow(len(r.SQL) + len(r.Params)*12) - - rest := r.SQL - for i, v := range r.Params { - q := strings.IndexByte(rest, '?') - if q < 0 { - return "", nil, fmt.Errorf("query has %d bound values but only %d placeholders", len(r.Params), i) - } - text, placeholder, err := chParamValue("p"+strconv.Itoa(i), v) - if err != nil { - return "", nil, err - } - b.WriteString(rest[:q]) - b.WriteString(placeholder) - params = append(params, text) - rest = rest[q+1:] - } - if strings.Contains(rest, "?") { - return "", nil, fmt.Errorf("query has more placeholders than the %d bound values", len(r.Params)) - } - b.WriteString(rest) - return b.String(), params, nil -} - -// chParamValue renders one bound value as its ClickHouse query-parameter text -// and the SQL its placeholder becomes, for the parameter called name. -func chParamValue(name string, v any) (text, placeholder string, err error) { - switch val := v.(type) { - case []any: - lit, err := chArrayLiteral(val) - if err != nil { - return "", "", err - } - return lit, "{" + name + ":Array(String)}", nil - case chsql.IntParam: - return chsql.EscapeStringParam(val.Value), chsql.StrictInt(name, val.Type), nil - } - raw, err := chScalarText(v) - if err != nil { - return "", "", err - } - return chsql.EscapeStringParam(raw), "{" + name + ":String}", nil -} - -// chScalarText is one scalar's value as plain text, before any encoding — -// what the caller means, not what the wire needs. -func chScalarText(v any) (string, error) { - switch val := v.(type) { - case string: - return val, nil - case json.Number: - // The caller's own digits, not a float64 round-trip: 12.50 stays - // "12.50" and an integer past 2^53 keeps every digit. - return val.String(), nil - case bool: - return strconv.FormatBool(val), nil - case float64: - return strconv.FormatFloat(val, 'f', -1, 64), nil - case int: - return strconv.Itoa(val), nil - case int64: - return strconv.FormatInt(val, 10), nil - case uint64: - return strconv.FormatUint(val, 10), nil - case nil: - // `col = NULL` is never true in SQL, so a null used to answer "no - // rows"; an empty String parameter would instead compare against the - // empty string, which is a different question. Refuse it (→ 400) - // rather than answer a question the caller did not ask. - return "", fmt.Errorf("filter value must not be null") - default: - return "", fmt.Errorf("unsupported filter value type %T", v) - } -} - -// chArrayLiteral renders a list as the `['a','b']` text an Array(String) -// query parameter is parsed from. A nested list has no place inside an `in` -// list, and the elements take quoteCHElement's encoding INSTEAD of -// chsql.EscapeStringParam's, not on top of it. -func chArrayLiteral(vals []any) (string, error) { - var b strings.Builder - b.WriteByte('[') - for i, v := range vals { - if _, isList := v.([]any); isList { - return "", fmt.Errorf("nested list in an 'in' value") - } - text, err := chScalarText(v) - if err != nil { - return "", err - } - if i > 0 { - b.WriteByte(',') - } - b.WriteString(quoteCHElement(text)) - } - b.WriteByte(']') - return b.String(), nil -} - -// quoteCHElement wraps one already-rendered value as a single-quoted element -// of an Array(String) parameter literal. That literal is read as a quoted -// value rather than an escaped field — a raw tab or newline inside the quotes -// round-trips untouched — so only the quote and the backslash need encoding, -// and chsql.EscapeStringParam's encoding must NOT be applied on top of it. -func quoteCHElement(s string) string { - var b strings.Builder - b.Grow(len(s) + 2) - b.WriteByte('\'') - for i := 0; i < len(s); i++ { - if s[i] == '\\' || s[i] == '\'' { - b.WriteByte('\\') - } - b.WriteByte(s[i]) - } - b.WriteByte('\'') - return b.String() -} diff --git a/internal/query/builder_test.go b/internal/query/builder_test.go index 0978c8da..ee525a2d 100644 --- a/internal/query/builder_test.go +++ b/internal/query/builder_test.go @@ -1,6 +1,7 @@ package query import ( + "encoding/binary" "encoding/json" "fmt" "strings" @@ -137,15 +138,45 @@ func TestBuild_InFilter(t *testing.T) { } result, err := Build("clicks", sq, testSchema(), nil, 0, DefaultMaxRows) require.NoError(t, err) - // One placeholder for the whole list, bound as one Array(String). + // One placeholder for the whole list, bound as one external table. assert.Contains(t, result.SQL, "`page` IN ?") require.Len(t, result.Params, 1) - assert.Equal(t, []any{"/home", "/about"}, result.Params[0]) + assert.Equal(t, listParam{Values: []any{"/home", "/about"}}, result.Params[0]) - sql, params, err := result.NamedParams() + b, err := result.Bind() require.NoError(t, err) - assert.Contains(t, sql, "`page` IN {p0:Array(String)}") - assert.Equal(t, []string{`['/home','/about']`}, params) + assert.Contains(t, b.SQL, "`page` IN (SELECT v FROM _p0)") + assert.Empty(t, b.Params) + assert.Equal(t, []Table{{Name: "_p0", Data: rowBinary("/home", "/about")}}, b.Tables) +} + +// rowBinary is the RowBinary a Table carries for vals. +func rowBinary(vals ...string) []byte { + var b []byte + for _, v := range vals { + b = binary.AppendUvarint(b, uint64(len(v))) + b = append(b, v...) + } + return b +} + +// bindTexts binds res and returns its SQL and parameter values, failing on a +// table: a test of scalar binding. +func bindTexts(t *testing.T, res *BuildResult) (string, []string) { + t.Helper() + b, err := res.Bind() + require.NoError(t, err) + require.Empty(t, b.Tables) + return b.SQL, paramValues(b) +} + +// paramValues are b's parameter values in placeholder order, nil for none. +func paramValues(b *Bound) []string { + var out []string + for _, p := range b.Params { + out = append(out, p.Value) + } + return out } func TestBuild_OrderBy(t *testing.T) { @@ -389,8 +420,7 @@ func TestBuild_PolicyPredicate_IntegerColumnsBindThroughTheStrictCast(t *testing perms := permsFiltering(map[string]policy.Filter{tt.column: tt.filter}, tt.claims) res, err := Build("clicks", &StructuredQuery{Columns: []string{"page"}}, schema, perms, 0, DefaultMaxRows) require.NoError(t, err) - sql, params, err := res.NamedParams() - require.NoError(t, err) + sql, params := bindTexts(t, res) assert.Equal(t, "SELECT `page` FROM `clicks` WHERE ("+tt.wantWhere+") LIMIT 10000", sql) assert.Equal(t, tt.wantParams, params) }) @@ -399,7 +429,8 @@ func TestBuild_PolicyPredicate_IntegerColumnsBindThroughTheStrictCast(t *testing // TestBuild_PolicyPredicate_CallerFiltersKeepThePlainForm: the strict cast is // for policy claims only. A caller's own filter on an integer column binds as -// before — it can only narrow what the policy already admits. +// a plain {pN:String}, and its `in` list as a table converted with +// accurateCastOrNull — it can only narrow what the policy already admits. func TestBuild_PolicyPredicate_CallerFiltersKeepThePlainForm(t *testing.T) { t.Parallel() perms := permsFiltering(map[string]policy.Filter{"count": {Eq: new("5")}}, nil) @@ -409,11 +440,12 @@ func TestBuild_PolicyPredicate_CallerFiltersKeepThePlainForm(t *testing.T) { }} res, err := Build("clicks", sq, testSchema(), perms, 0, DefaultMaxRows) require.NoError(t, err) - sql, params, err := res.NamedParams() + b, err := res.Bind() require.NoError(t, err) assert.Equal(t, "SELECT `page` FROM `clicks` WHERE (`count` = "+chsql.StrictInt("p0", "UInt64")+ - ") AND `count` > {p1:String} AND `count` IN {p2:Array(String)} LIMIT 10000", sql) - assert.Equal(t, []string{"5", "1", "['1','2']"}, params) + ") AND `count` > {p1:String} AND `count` IN (SELECT accurateCastOrNull(v, 'UInt64') FROM _p2) LIMIT 10000", b.SQL) + assert.Equal(t, []string{"5", "1"}, paramValues(b)) + assert.Equal(t, []Table{{Name: "_p2", Data: rowBinary("1", "2")}}, b.Tables) } // TestBuild_PolicyMaxRows pins the role's max_rows cap folded into Build's LIMIT @@ -568,9 +600,10 @@ func TestResolveTimeValue_RelativeDuration(t *testing.T) { require.NoError(t, err) assert.NotEmpty(t, result) assert.NotEqual(t, "1h", result, "relative duration should resolve to a timestamp") - - assert.NotContains(t, result, "T", "must not emit the RFC3339 T separator") - assert.NotContains(t, result, "Z", "must not emit the RFC3339 Z zone suffix") + // RFC 3339 in UTC: the Z is what makes the column's zone irrelevant. + _, err = time.Parse(time.RFC3339Nano, result) + require.NoError(t, err) + assert.True(t, strings.HasSuffix(result, "Z"), result) } func TestResolveTimeValue_RFC3339(t *testing.T) { @@ -578,7 +611,12 @@ func TestResolveTimeValue_RFC3339(t *testing.T) { result, err := resolveTimeValue("2024-01-01T00:00:00Z", 0) require.NoError(t, err) - assert.Equal(t, "2024-01-01 00:00:00", result) + assert.Equal(t, "2024-01-01T00:00:00Z", result) + + // An offset is normalised to the same instant in UTC; a fraction is kept. + result, err = resolveTimeValue("2024-01-01T09:00:00.25+09:00", 0) + require.NoError(t, err) + assert.Equal(t, "2024-01-01T00:00:00.25Z", result) } func TestResolveTimeValue_WithBucketing(t *testing.T) { @@ -586,7 +624,7 @@ func TestResolveTimeValue_WithBucketing(t *testing.T) { // With 60s buckets, a time at :30 should truncate to :00. result, err := resolveTimeValue("2024-01-01T12:34:30Z", 60) require.NoError(t, err) - assert.Equal(t, "2024-01-01 12:34:00", result) + assert.Equal(t, "2024-01-01T12:34:00Z", result) } func TestExpandDayWeek(t *testing.T) { @@ -610,13 +648,13 @@ func TestExpandDayWeek(t *testing.T) { func TestResolveTimeValue_DayWeekSuffix(t *testing.T) { t.Parallel() // "7d"/"2w" are documented (sdk.md) but not Go durations; they must resolve - // to a real ClickHouse DateTime, not fall through as the raw string (#285). + // to an instant, not fall through as the raw string (#285). for _, in := range []string{"7d", "2w", "1d12h"} { result, err := resolveTimeValue(in, 0) require.NoError(t, err, "resolveTimeValue(%q)", in) assert.NotEqual(t, in, result, "%q must resolve, not pass through raw", in) - assert.NotContains(t, result, "T") - assert.NotContains(t, result, "Z") + _, err = time.Parse(time.RFC3339Nano, result) + require.NoError(t, err, "resolveTimeValue(%q) = %q", in, result) } // "7d" must resolve to the same instant as its hour-equivalent "168h". @@ -647,30 +685,41 @@ func TestBucketTime_ZeroBucket(t *testing.T) { assert.Equal(t, ts, got, "zero bucket should not truncate") } -func TestCoerceFilterValue(t *testing.T) { +// TestConversionFor pins how a column's type, as system.columns spells it, +// picks the conversion ClickHouse applies to a bound String. +func TestConversionFor(t *testing.T) { t.Parallel() - tests := []struct { - name string - input any - wantTyp string - wantVal any + typ string + want conversion }{ - {"RFC3339", "2026-04-02T16:02:07Z", "string", "2026-04-02 16:02:07"}, - {"RFC3339Nano", "2026-04-02T16:02:07.666Z", "string", "2026-04-02 16:02:07.666"}, - {"RFC3339Nano_short", "2026-04-02T16:02:07.15Z", "string", "2026-04-02 16:02:07.15"}, - {"plain_string", "hello", "string", "hello"}, - {"number", 42, "int", 42}, - {"nil", nil, "", nil}, + {"DateTime", conversion{kind: parseTime, scale: 8}}, + {"DateTime('Asia/Tokyo')", conversion{kind: parseTime, scale: 8, zone: "Asia/Tokyo"}}, + {"DateTime64(3)", conversion{kind: parseTime, scale: 8}}, + {"DateTime64(3, 'America/New_York')", conversion{kind: parseTime, scale: 8, zone: "America/New_York"}}, + {"DateTime64(9, 'UTC')", conversion{kind: parseTime, scale: 9, zone: "UTC"}}, + {"Nullable(DateTime64(6, 'Etc/GMT+5'))", conversion{kind: parseTime, scale: 8, zone: "Etc/GMT+5"}}, + {"LowCardinality(Nullable(DateTime('America/Argentina/Buenos_Aires')))", conversion{kind: parseTime, scale: 8, zone: "America/Argentina/Buenos_Aires"}}, + {"Date", conversion{kind: parseDate}}, + {"Nullable(Date32)", conversion{kind: parseDate32}}, + {"String", conversion{kind: asString}}, + {"LowCardinality(String)", conversion{kind: asString}}, + {"", conversion{kind: asString}}, + {"UInt64", conversion{kind: castTo, typ: "UInt64"}}, + {"LowCardinality(Nullable(Int32))", conversion{kind: castTo, typ: "Int32"}}, + {"Decimal(18, 4)", conversion{kind: castTo, typ: "Decimal(18, 4)"}}, + {"Enum8('a' = 1, 'b' = 2)", conversion{kind: castTo, typ: "Enum8('a' = 1, 'b' = 2)"}}, + {"Array(DateTime)", conversion{kind: castTo, typ: "Array(DateTime)"}}, + // A DateTime type it cannot read leaves ClickHouse to refuse what it + // cannot compare, rather than guess the zone. + {"DateTime64(x)", conversion{kind: asString}}, + {"DateTime('a\\'b')", conversion{kind: asString}}, + {"DateTime64(3, 'UTC', 1)", conversion{kind: asString}}, } for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { + t.Run(tt.typ, func(t *testing.T) { t.Parallel() - got := coerceFilterValue(tt.input) - assert.Equal(t, tt.wantTyp, fmt.Sprintf("%T", got)) - if tt.wantVal != nil { - assert.Equal(t, tt.wantVal, got) - } + assert.Equal(t, tt.want, conversionFor(tt.typ)) }) } } @@ -695,60 +744,72 @@ func TestBuild_FilterWithTimestampValue(t *testing.T) { result, err := Build("events", sq, schema, nil, 0, DefaultMaxRows) require.NoError(t, err) assert.Contains(t, result.SQL, "`received_timestamp` < ?") - require.Len(t, result.Params, 1) - strVal, isString := result.Params[0].(string) - assert.True(t, isString, "timestamp filter value should be coerced to formatted string, got %T", result.Params[0]) - assert.Equal(t, "2026-04-02 16:02:07.666", strVal) - sql, params, err := result.NamedParams() - require.NoError(t, err) - assert.Contains(t, sql, "`received_timestamp` < {p0:String}") - assert.Equal(t, []string{"2026-04-02 16:02:07.666"}, params) + sql, params := bindTexts(t, result) + assert.Contains(t, sql, "`received_timestamp` < parseDateTime64BestEffort({p0:String}, 8)") + assert.Equal(t, []string{"2026-04-02T16:02:07.666Z"}, params, "the value reaches ClickHouse as written") } -// TestBuild_TimestampRewriteIsForDateTimeColumnsOnly: an RFC3339 value is -// rewritten only where the schema types the column DateTime or DateTime64 — -// wrapped or not, scalar or inside an `in` list. The same text on any other -// column is the caller's own value and reaches ClickHouse as written. -func TestBuild_TimestampRewriteIsForDateTimeColumnsOnly(t *testing.T) { +// TestBuild_TimestampValuesParseInClickHouse: a filter value on a Date or +// DateTime column — wrapped or not, scalar or inside an `in` list — reaches +// ClickHouse as written and is parsed there, in the column's declared zone +// when it has one. The same text on any other column is compared as it is, +// and like never converts. +func TestBuild_TimestampValuesParseInClickHouse(t *testing.T) { t.Parallel() schema := &discovery.TableSchema{Name: "events", Columns: []discovery.Column{ {Name: "dt", Type: "DateTime"}, {Name: "dtz", Type: "DateTime('Europe/Berlin')"}, {Name: "dt64", Type: "Nullable(DateTime64(3, 'UTC'))"}, + {Name: "dt9", Type: "DateTime64(9, 'Asia/Tokyo')"}, {Name: "lc", Type: "LowCardinality(Nullable(DateTime))"}, {Name: "label", Type: "String"}, {Name: "day", Type: "Date"}, + {Name: "day32", Type: "Date32"}, }} const rfc = "2026-04-02T16:02:07.666Z" - const ch = "2026-04-02 16:02:07.666" tests := []struct { - column string - value any - want any + column, op string + value any + wantWhere string + wantParams []string + wantTable []byte }{ - {"dt", rfc, ch}, - {"dtz", rfc, ch}, - {"dt64", rfc, ch}, - {"lc", rfc, ch}, - {"dt", "2026-04-02 16:02:07", "2026-04-02 16:02:07"}, - {"label", rfc, rfc}, - {"day", rfc, rfc}, - {"dt64", []any{rfc, "2026-04-02 16:02:07"}, []any{ch, "2026-04-02 16:02:07"}}, - {"label", []any{rfc}, []any{rfc}}, + {"dt", "eq", rfc, "`dt` = parseDateTime64BestEffort({p0:String}, 8)", []string{rfc}, nil}, + {"dtz", "gte", rfc, "`dtz` >= parseDateTime64BestEffort({p0:String}, 8, 'Europe/Berlin')", []string{rfc}, nil}, + {"dt64", "neq", rfc, "`dt64` != parseDateTime64BestEffort({p0:String}, 8, 'UTC')", []string{rfc}, nil}, + {"dt9", "lt", rfc, "`dt9` < parseDateTime64BestEffort({p0:String}, 9, 'Asia/Tokyo')", []string{rfc}, nil}, + {"lc", "lte", rfc, "`lc` <= parseDateTime64BestEffort({p0:String}, 8)", []string{rfc}, nil}, + {"dtz", "eq", "2026-04-02 16:02:07", "`dtz` = parseDateTime64BestEffort({p0:String}, 8, 'Europe/Berlin')", []string{"2026-04-02 16:02:07"}, nil}, + {"dt", "gt", json.Number("1782014400"), "`dt` > parseDateTime64BestEffort({p0:String}, 8)", []string{"1782014400"}, nil}, + {"day", "eq", rfc, "`day` = toDate(parseDateTime64BestEffort({p0:String}, 8))", []string{rfc}, nil}, + {"day32", "eq", "1950-01-01", "`day32` = toDate32(parseDateTime64BestEffort({p0:String}, 8))", []string{"1950-01-01"}, nil}, + {"label", "eq", rfc, "`label` = {p0:String}", []string{rfc}, nil}, + {"dt", "like", "2026-%", "`dt` LIKE {p0:String}", []string{"2026-%"}, nil}, + { + "dt64", "in", + []any{rfc, "2026-04-02 16:02:07"}, + "`dt64` IN (SELECT parseDateTime64BestEffort(v, 8, 'UTC') FROM _p0)", nil, + rowBinary(rfc, "2026-04-02 16:02:07"), + }, + {"day", "in", []any{"2026-04-02"}, "`day` IN (SELECT toDate(parseDateTime64BestEffort(v, 8)) FROM _p0)", nil, rowBinary("2026-04-02")}, + {"label", "in", []any{rfc}, "`label` IN (SELECT v FROM _p0)", nil, rowBinary(rfc)}, } for _, tt := range tests { - t.Run(fmt.Sprintf("%s %v", tt.column, tt.value), func(t *testing.T) { + t.Run(fmt.Sprintf("%s %s %v", tt.column, tt.op, tt.value), func(t *testing.T) { t.Parallel() - op := "eq" - if _, ok := tt.value.([]any); ok { - op = "in" - } - sq := &StructuredQuery{SelectAll: true, Filters: []Filter{{Column: tt.column, Op: op, Value: tt.value}}} + sq := &StructuredQuery{Columns: []string{"label"}, Filters: []Filter{{Column: tt.column, Op: tt.op, Value: tt.value}}} result, err := Build("events", sq, schema, nil, 0, DefaultMaxRows) require.NoError(t, err) - require.Len(t, result.Params, 1) - assert.Equal(t, tt.want, result.Params[0]) + b, err := result.Bind() + require.NoError(t, err) + assert.Equal(t, "SELECT `label` FROM `events` WHERE "+tt.wantWhere+" LIMIT 10000", b.SQL) + assert.Equal(t, tt.wantParams, paramValues(b)) + if tt.wantTable == nil { + assert.Empty(t, b.Tables) + } else { + assert.Equal(t, []Table{{Name: "_p0", Data: tt.wantTable}}, b.Tables) + } }) } } @@ -830,26 +891,29 @@ func TestBuild_TimeRange_SinceOnly(t *testing.T) { assert.NotContains(t, result.SQL, "`ts` <= ?") } -func TestBuild_TimeRange_ClickHouseDateTimeFormat(t *testing.T) { +// TestBuild_TimeRange_BoundsAreRFC3339: time_range bounds are instants in +// UTC, bound through the column's conversion, so the column's zone cannot +// shift them (#238 was a T-separated bound ClickHouse refused). +func TestBuild_TimeRange_BoundsAreRFC3339(t *testing.T) { t.Parallel() + schema := &discovery.TableSchema{Name: "clicks", Columns: []discovery.Column{ + {Name: "page", Type: "String"}, + {Name: "ts", Type: "DateTime('Asia/Tokyo')"}, + }} sq := &StructuredQuery{ Columns: []string{"page"}, TimeRange: &TimeRange{ Column: "ts", - Since: "2024-01-01T00:00:00Z", + Since: "2024-01-01T09:00:00+09:00", Until: "2024-01-02T03:04:05Z", }, } - result, err := Build("clicks", sq, testSchema(), nil, 0, DefaultMaxRows) + result, err := Build("clicks", sq, schema, nil, 0, DefaultMaxRows) require.NoError(t, err) - require.Len(t, result.Params, 2) - assert.Equal(t, "2024-01-01 00:00:00", result.Params[0]) - assert.Equal(t, "2024-01-02 03:04:05", result.Params[1]) - for _, p := range result.Params { - s, ok := p.(string) - require.True(t, ok, "time_range bound must be a formatted string param") - assert.NotContains(t, s, "T", "must not bind an RFC3339 T-separated string (#238)") - } + sql, params := bindTexts(t, result) + assert.Equal(t, "SELECT `page` FROM `clicks` WHERE `ts` >= parseDateTime64BestEffort({p0:String}, 8, 'Asia/Tokyo')"+ + " AND `ts` <= parseDateTime64BestEffort({p1:String}, 8, 'Asia/Tokyo') LIMIT 10000", sql) + assert.Equal(t, []string{"2024-01-01T00:00:00Z", "2024-01-02T03:04:05Z"}, params) } func TestBuild_FilterUnsupportedOp(t *testing.T) { @@ -1212,37 +1276,52 @@ func TestBuild_InsertResolvedGrantIsRejected(t *testing.T) { assert.NotNil(t, res) } -// TestNamedParams pins the ClickHouse binding rule: every positional `?` -// becomes a named parameter, every scalar binds as String and every list as -// Array(String), in the order the WHERE assembly emitted them. -func TestNamedParams(t *testing.T) { +// TestBind pins the ClickHouse binding rule: every positional `?` becomes a +// named parameter or an external table, every scalar binds as String and +// every list as a table, in the order the WHERE assembly emitted them. +func TestBind(t *testing.T) { t.Parallel() tests := []struct { name string sql string params []any wantSQL string - wantParams []string + wantParams []Param + wantTables []Table }{ { - name: "no parameters", - sql: "SELECT `page` FROM `clicks` LIMIT 10", - wantSQL: "SELECT `page` FROM `clicks` LIMIT 10", - wantParams: nil, + name: "no parameters", + sql: "SELECT `page` FROM `clicks` LIMIT 10", + wantSQL: "SELECT `page` FROM `clicks` LIMIT 10", }, { name: "policy predicate keeps its leading position", sql: "SELECT `page` FROM `clicks` WHERE (`org_id` = ?) AND `page` = ? LIMIT 100", params: []any{"org-1", "/home"}, wantSQL: "SELECT `page` FROM `clicks` WHERE (`org_id` = {p0:String}) AND `page` = {p1:String} LIMIT 100", - wantParams: []string{"org-1", "/home"}, + wantParams: []Param{{"p0", "org-1"}, {"p1", "/home"}}, }, { - name: "list binds as one Array(String)", - sql: "SELECT * FROM `t` WHERE `page` IN ? LIMIT 10", - params: []any{[]any{"/a", "/b"}}, - wantSQL: "SELECT * FROM `t` WHERE `page` IN {p0:Array(String)} LIMIT 10", - wantParams: []string{`['/a','/b']`}, + name: "list binds as one table, named after its position", + sql: "SELECT * FROM `t` WHERE `a` = ? AND `page` IN ? AND `b` = ? LIMIT 10", + params: []any{"x", listParam{Values: []any{"/a", "/b"}}, "y"}, + wantSQL: "SELECT * FROM `t` WHERE `a` = {p0:String} AND `page` IN (SELECT v FROM _p1) AND `b` = {p2:String} LIMIT 10", + wantParams: []Param{{"p0", "x"}, {"p2", "y"}}, + wantTables: []Table{{Name: "_p1", Data: rowBinary("/a", "/b")}}, + }, + { + name: "a list on a typed column is cast element by element", + sql: "SELECT * FROM `t` WHERE `e` IN ? LIMIT 10", + params: []any{listParam{Values: []any{"it's"}, Conv: conversionFor("Enum8('it\\'s' = 1)")}}, + wantSQL: "SELECT * FROM `t` WHERE `e` IN (SELECT accurateCastOrNull(v, 'Enum8(\\'it\\\\\\'s\\' = 1)') FROM _p0) LIMIT 10", + wantTables: []Table{{Name: "_p0", Data: rowBinary("it's")}}, + }, + { + name: "a timestamp parses in ClickHouse", + sql: "SELECT * FROM `t` WHERE `ts` > ? LIMIT 10", + params: []any{conversionFor("DateTime('Asia/Tokyo')").scalar("2026-06-21T04:00:00.5Z")}, + wantSQL: "SELECT * FROM `t` WHERE `ts` > parseDateTime64BestEffort({p0:String}, 8, 'Asia/Tokyo') LIMIT 10", + wantParams: []Param{{"p0", "2026-06-21T04:00:00.5Z"}}, }, { name: "numbers keep the caller's own digits", @@ -1250,37 +1329,31 @@ func TestNamedParams(t *testing.T) { params: []any{json.Number("12.50"), json.Number("9007199254740993"), true}, wantSQL: "SELECT * FROM `t` WHERE `a` = {p0:String} AND `b` = {p1:String} " + "AND `c` = {p2:String} LIMIT 10", - wantParams: []string{"12.50", "9007199254740993", "true"}, + wantParams: []Param{{"p0", "12.50"}, {"p1", "9007199254740993"}, {"p2", "true"}}, }, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { t.Parallel() - sql, params, err := (&BuildResult{SQL: tt.sql, Params: tt.params}).NamedParams() + b, err := (&BuildResult{SQL: tt.sql, Params: tt.params}).Bind() require.NoError(t, err) - assert.Equal(t, tt.wantSQL, sql) - assert.Equal(t, tt.wantParams, params) + assert.Equal(t, tt.wantSQL, b.SQL) + assert.Equal(t, tt.wantParams, b.Params) + assert.Equal(t, tt.wantTables, b.Tables) }) } } -// TestNamedParams_Encoding pins the two escapings a ClickHouse query parameter -// needs, which are NOT the same and must not be applied to each other's -// values. Measured on 26.6.3.62 (the version the integration suite pins): -// -// - a scalar `{p:String}` is read by the escaped-text reader, so a raw -// backslash is taken as the start of an escape sequence ("a\b" came back -// holding a backspace) and a raw tab or newline ends the field outright -// (code 457, a 500 for the caller); -// - an `Array(String)` value is an array literal whose elements are quoted, -// so a raw tab or newline rides through untouched and only the quote and -// the backslash need encoding. -// -// Both encodings round-trip every case below byte for byte against a live -// server; getting either wrong is silent data loss, not an error. -func TestNamedParams_Encoding(t *testing.T) { +// TestBind_Encoding pins how a value travels. A scalar {p:String} is read by +// ClickHouse's escaped-text reader, so a raw backslash is taken as the start +// of an escape sequence ("a\b" came back holding a backspace, measured on +// 26.6.3.62) and a raw tab ends the field outright (code 457, measured again +// on 26.8.15.10); chsql.EscapeStringParam encodes both. An `in` list's +// elements travel as RowBinary, each one's bytes as written behind its +// length, so nothing in one can end it or reach the SQL. +func TestBind_Encoding(t *testing.T) { t.Parallel() - tests := []struct { + scalars := []struct { name string value any wantParam string @@ -1294,25 +1367,44 @@ func TestNamedParams_Encoding(t *testing.T) { {"carriage return", "a\rb", `a\rb`}, {"a literal backslash-n", `a\nb`, `a\\nb`}, {"like pattern", "%foo%", "%foo%"}, - {"list quotes and backslashes", []any{`it's`, `a\b`, "a\tb"}, "['it\\'s','a\\\\b','a\tb']"}, - {"list containment attempt", []any{`']) OR 1=1 --`}, `['\']) OR 1=1 --']`}, } - for _, tt := range tests { + for _, tt := range scalars { t.Run(tt.name, func(t *testing.T) { t.Parallel() - _, params, err := (&BuildResult{SQL: "SELECT ?", Params: []any{tt.value}}).NamedParams() - require.NoError(t, err) - require.Len(t, params, 1) - assert.Equal(t, tt.wantParam, params[0]) + _, params := bindTexts(t, &BuildResult{SQL: "SELECT ?", Params: []any{tt.value}}) + assert.Equal(t, []string{tt.wantParam}, params) }) } + + t.Run("list elements are their own bytes", func(t *testing.T) { + t.Parallel() + vals := []string{`it's`, `a\b`, "a\tb", "a\nb", `']) OR 1=1 --`, "", "\x00z", strings.Repeat("x", 300)} + list := make([]any, len(vals)) + for i, v := range vals { + list[i] = v + } + b, err := (&BuildResult{SQL: "SELECT * FROM t WHERE c IN ?", Params: []any{listParam{Values: list}}}).Bind() + require.NoError(t, err) + assert.Equal(t, "SELECT * FROM t WHERE c IN (SELECT v FROM _p0)", b.SQL) + require.Len(t, b.Tables, 1) + // Decode the RowBinary back: varint length, then the bytes. + var got []string + for data := b.Tables[0].Data; len(data) > 0; { + n, w := binary.Uvarint(data) + require.Positive(t, w) + end := w + int(n) //nolint:gosec // G115: a test value's length, far below MaxInt + got = append(got, string(data[w:end])) + data = data[end:] + } + assert.Equal(t, vals, got) + }) } -// TestNamedParams_Rejects covers the values and shapes that have no honest -// binding. A JSON null is the notable one: it used to bind as `col = NULL` -// (never true), where an empty String parameter would compare -// against the empty string — a different question, so it is refused (→ 400). -func TestNamedParams_Rejects(t *testing.T) { +// TestBind_Rejects covers the values and shapes that have no honest binding. +// A JSON null is the notable one: it used to bind as `col = NULL` (never +// true), where an empty String parameter would compare against the empty +// string — a different question, so it is refused (→ 400). +func TestBind_Rejects(t *testing.T) { t.Parallel() tests := []struct { name string @@ -1321,8 +1413,11 @@ func TestNamedParams_Rejects(t *testing.T) { wantErr string }{ {"null value", "SELECT ?", []any{nil}, "must not be null"}, + {"null in a list", "SELECT ?", []any{listParam{Values: []any{"a", nil}}}, "must not be null"}, + {"null timestamp", "SELECT ?", []any{conversionFor("DateTime").scalar(nil)}, "must not be null"}, {"object value", "SELECT ?", []any{map[string]any{"k": "v"}}, "unsupported filter value type"}, - {"nested list", "SELECT ?", []any{[]any{[]any{"a"}}}, "nested list"}, + {"list as a scalar", "SELECT ?", []any{[]any{"a"}}, "unsupported filter value type"}, + {"nested list", "SELECT ?", []any{listParam{Values: []any{[]any{"a"}}}}, "nested list"}, {"more values than placeholders", "SELECT 1", []any{"a"}, "only 0 placeholders"}, {"more placeholders than values", "SELECT ?, ?", []any{"a"}, "more placeholders"}, {"placeholder with no values", "SELECT ?", nil, "no bound values"}, @@ -1330,7 +1425,7 @@ func TestNamedParams_Rejects(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { t.Parallel() - _, _, err := (&BuildResult{SQL: tt.sql, Params: tt.params}).NamedParams() + _, err := (&BuildResult{SQL: tt.sql, Params: tt.params}).Bind() require.Error(t, err) assert.Contains(t, err.Error(), tt.wantErr) }) From 3858fb7314972ab86038d831b36024f644a38e91 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:43:35 -0400 Subject: [PATCH 22/70] docs(query): timestamp filters parse in ClickHouse; in lists are uncapped The api and architecture pages, and the Unreleased read-path bullet, said a DateTime filter value went to ClickHouse verbatim and an in list was one Array(String) parameter with a 400 past its size. Both changed: say what the builder sends now, the limits that remain and their 400s, and add a Fixed entry for the zone bug. The e2e query suite gains a DateTime('Asia/Tokyo') column (RFC 3339, offset and zone-less values on the same row), a fractional bound on a whole-second column, and in lists of 20,000 and 30,000 elements, past ClickHouse's 128 KiB parameter cap. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- CHANGELOG.md | 4 +++- docs/src/content/docs/api.md | 11 +++++---- docs/src/content/docs/architecture.md | 3 ++- tests/e2e/sdk/query.test.ts | 33 +++++++++++++++++++++------ 4 files changed, 37 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 8eb238cb..761743bf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -93,7 +93,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The landing page's live demo now reads from the stats deployment's new WaveHouse Cloud backend** (`docs/src/components/LiveDemo.astro`, `docs/scripts/screenshot.mjs`): the GitHub-activity dogfood deployment behind the hero panel (Wave-RF/WaveHouse-Stats) moved off its self-managed AWS infrastructure onto WaveHouse Cloud, so `BASE_URL` — the origin `@wavehouse/sdk` queries in the visitor's browser — points at `https://iefrrvavd5akvphk7pq3.wavehouse.app` instead of `https://stats.wavehouse.dev`, ahead of the AWS stack being torn down. The `PUBLIC_WAVEHOUSE_STATS_URL` build-time override is unchanged, so a fork or staging docs build still redirects the panel without a code edit. **`DEMO_HOST` deliberately stays `stats.wavehouse.dev`** — the demo *site* is still served there and is still what the panel's chrome label and "Full demo" link should show; the migration splits the site from the API origin behind it, and the two constants now carry comments saying so. Verified against the new deployment before the switch: all five pipes the panel reads (`gh_summary`, `gh_activity_recent`, `gh_events_per_minute`, and the pre-#19 `gh_stars_total` / `gh_forks_total` fallbacks) return `200` with the same row shapes, the structured-query backfill fallback (`POST /v1/query?table=gh_events`) matches its old-backend response byte for byte, `GET /v1/stream?table=gh_events` opens an SSE stream, and CORS is unchanged (`Access-Control-Allow-Origin: *`, `X-Cache` exposed) so the cross-origin browser reads keep working from the docs site. The new backend is already the live ingest target — it reported more recent events than the old one at cutover (5,175 vs 5,038 over 7d) — which is the other half of why the panel had to follow it. `screenshot.mjs`'s `networkidle` note is retargeted to "the stats demo backend" rather than naming a host it no longer connects to. -- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is ClickHouse's spelling (`"2026-06-21 04:00:00.123"`, in the column's zone, no offset) where it was RFC 3339 UTC, and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list binds as one `Array(String)` parameter, with a size and count check that answers `400` before the request is sent. The RFC 3339 filter-value rewrite is deleted, so a `"…Z"` value goes to the server verbatim and works as far as the connected server's version reads that spelling. Failure classification is unchanged — the same `code`/`retryable` table of the query paths — and each tenant's reader connections are capped at its `max_open_conns`. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. +- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is ClickHouse's spelling (`"2026-06-21 04:00:00.123"`, in the column's zone, no offset) where it was RFC 3339 UTC, and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification is unchanged — the same `code`/`retryable` table of the query paths — and each tenant's reader connections are capped at its `max_open_conns`. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. - **WH001 (no hard-wrapped prose) now applies to every tracked Markdown file, with no carve-out** (`.github/.markdownlint.json` (deleted), `.claude/.markdownlint.json` (deleted), `.claude/skills/integration-astro-view-transitions/` (deleted), `.markdownlint-cli2.jsonc`, `.github/workflows/README.md`, `.claude/skills/pm-triage/references/routine.md`, `AGENTS.md`, `scripts/docs-prose.sh`, `.github/prompts/docs-review.md`, `docs/src/content/docs/claude-code.md`, `docs/src/content/docs/development.md`, `.claude/agents/docs-reviewer.md`): two path-scoped configs had switched WH001 off under `.github/` and `.claude/` ever since [#489](https://github.com/Wave-RF/WaveHouse/pull/489) introduced the rule — baked in from the start rather than added in response to a discovered problem — which left the repo documenting the rule three ways and disagreeing with itself: `CONTRIBUTING.md` promises contributors `make lint` enforces it *everywhere*, while `AGENTS.md` and the `.markdownlint-cli2.jsonc` header wrote up the carve-out. Not theoretical: on [#520](https://github.com/Wave-RF/WaveHouse/pull/520) a reviewer correctly flagged a hard-wrapped bullet in `.github/workflows/README.md`, an agent pointed at `"WH001": false` for that path and pushed back, and the reviewer recorded a *learning* never to flag WH001 there — the wrong invariant, learned off the wrong side of the contradiction ([#521](https://github.com/Wave-RF/WaveHouse/issues/521)). Both configs are deleted — each held nothing but the override, so the root `.markdownlint.json` governs again — and the 51 hard-wrapped paragraphs they were hiding are joined: 41 in `.github/workflows/README.md` and 10 in `.claude/skills/pm-triage/references/routine.md`, mechanical joins with no wording changed and every fenced block, table row, and heading byte-identical either side of the reflow. Deleted with them: the wizard-installed PostHog skill at `.claude/skills/integration-astro-view-transitions/` — 9 files, ~1,456 lines, including an 809-line `EXAMPLE.md` copied wholesale from `PostHog/context-mill`. Its integration job finished in [#277](https://github.com/Wave-RF/WaveHouse/pull/277), nothing in the repo calls it, and the docs-site setup it once described is documented where it belongs — in `docs/src/components/PostHog.astro` and this file. Keeping unowned third-party prose in the tree means content that drifts silently on every upstream bump and that nobody here reviews; it was also the single file that would have needed a special-case lint exclusion, so removing it is what lets WH001 apply with **no exception at all** rather than one documented one. Its two inventory rows in `claude-code.md` go with it, as does the now-dead `docs/posthog-setup-report.md` entry in the `scripts/docs-prose.sh` denylist (the wizard's other artifact, deleted back in [#502](https://github.com/Wave-RF/WaveHouse/pull/502)) and the copies of that denylist in `AGENTS.md` and `.github/prompts/docs-review.md`, which the script's header requires be kept in lockstep. Review of the change then turned up four more things the exclusion had been hiding, all fixed here: **WH001 has a blind spot** — `no-hard-wrapped-prose.mjs` classifies any line indented four or more spaces as an indented code block, so a *nested* list item is never joined, which left three hard-wrapped bullets in `.github/workflows/README.md` §"Adding a job" that the autofix could not see (unwrapped by hand; they were the last hard-wrapped prose paragraphs in the repo) and made `AGENTS.md`'s and `development.md`'s "a list item is joined as a unit" wrong for nested items (both now state the four-space caveat); the `scripts/docs-prose.sh` header told readers to keep its denylist in lockstep with **two** sibling copies when there are **three** — the missed one being `.claude/agents/docs-reviewer.md`, the gating subagent's own system prompt, which had in fact been silently out of sync for the whole life of the `posthog-setup-report.md` exclusion; the `.markdownlint-cli2.jsonc` header's "applies to every tracked Markdown file" was exact for WH001 but not WH002, which returns early on anything that isn't `.mdx`; and the job-graph diagram omitted `docs-deploy`'s `needs` edges from `unit`, `integration`, and `e2e`, contradicting invariant 2 three lines below it. The denylist also drops its `PERF-CLAIMS-REVIEW.md` entry — unlike the wizard artifact this one names a file that was **never tracked** at all, so it guarded a hypothetical; the list's other general cases are patterns (`*.draft.md`, `*.old.md`) that already cover a one-off review document, and a literal filename restated in four places is the outlier. `scripts/docs-prose.sh all` still resolves the same 27-file prose set. @@ -109,6 +109,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Fixed +- **A timestamp filter on a column with a time zone matches the instant it names** (`internal/query/{builder,bind}.go` (`bind.go` new; + tests), `internal/api/{clickhouse_http,structured_query}.go` (+ tests, + integration-tagged `structured_query_integration_test.go`), `tests/e2e/sdk/query.test.ts`, `docs/src/content/docs/{api,architecture}.md`): `POST /v1/query` rewrote an RFC 3339 filter value, and formatted `time_range` bounds, as zone-less UTC text that ClickHouse then read in the **column's** zone — so on a `DateTime('Asia/Tokyo')` column `eq "2026-06-21T04:00:00Z"` matched nothing, on `DateTime64(3, 'America/New_York')` it matched the row four hours later, and on a zone-less column under a non-UTC server the row the server's offset away (measured on ClickHouse 26.8.15.10 and 24.8.14.39). WaveHouse no longer rewrites the value: on a `Date`/`DateTime`/`DateTime64` column it binds as written and ClickHouse parses it with `parseDateTime64BestEffort` in the column's declared zone, so an offset or `Z` is the exact instant, a zone-less value still means the column's local time, a fraction compares exactly against a whole-second column (ClickHouse refused it as a type mismatch before), and a value ClickHouse cannot parse is `400 clickhouse.rejected`. The primary key is still used. `time_range` bounds go as RFC 3339 in UTC through the same parse. **An `in` list is no longer capped**: it went as one query-string parameter, which ClickHouse caps at 128 KiB, and answered `400` past that; it now travels as an external table in a multipart body, so any list the 1 MiB request carries reaches ClickHouse, compared under the column's type and with the primary key in use. One difference from the old list: an element that is not a value of the column's type (`"1.5"` on an integer column) matches no row instead of failing the query. The limits left are ClickHouse's own, each a `400` that names it: one scalar value over 128 KiB once URL-encoded, all of them over the request line, and — for a query with an `in` list — SQL over the 128 KiB form field it travels in. + - **A tenant moved to another ClickHouse address or database discovers its new schema with the reload** (`internal/app/{wire,discoveries}.go` (+ tests), `internal/discovery/discovery.go` (comment), `internal/testutil/testutil.go`, `docs/src/content/docs/architecture.md`, `settings-directory.mdx`, `AGENTS.md`): closes [#638](https://github.com/Wave-RF/WaveHouse/issues/638). A reload that changed a tenant's `clickhouse.addr` or `clickhouse.database` orphaned the tenant's cache and left its schema registry as it was, so until the tenant's loop fired at `schema.refresh_interval` its queries and inserts were validated against the previous database's schema and run against the new one. The tenants the pools reconcile reports stale now have their registry dropped in the same hook, and the discovery reconcile that follows builds each a fresh one over the pool it is on, as it does for a tenant back after a rejection or removal: the first discovery runs at once in the tenant's own loop, so the reload never waits on ClickHouse, and until it succeeds the tenant's table lookups answer `503` with `Retry-After: 5`, as before any first discovery. A discovery that fails is logged with its tenant (`schema discovery retry failed`), counted in `wavehouse_schema_refresh_failures_total`, and retried with backoff from two seconds to sixty. A tenant whose address and database did not change keeps its registry and its loop, a flat directory's tenant `0` moves the same way, and a process without the api role, which discovers no schema, only repoints its pool. - **A shard's owner keeps it while its ClickHouse is slow, one stuck shard no longer stalls the process's others, and a handover or clean stop keeps each table's rows in order** (`internal/mq/external_consumer.go` (new, from `external.go`), `internal/mq/{external,mq}.go`, `internal/ingest/{claims,worker}.go`, `.testcoverage.yml`, tests in `internal/mq/external_test.go`, `internal/ingest/{claims,worker}_test.go` and `tests/integration/shard_order_test.go` (new), `docs/src/content/docs/{deployment,ingest-pipeline,architecture}.md`, `AGENTS.md`): part of the external-NATS workstream of [#613](https://github.com/Wave-RF/WaveHouse/issues/613), fixing the shard ownership of the entries above before they ship. The pull consumer ran the worker's handler on its delivery goroutine and pulled again only when it returned, and the server renews a shard's pin only on a pull, so a process at its cap of held rows for longer than the pinned TTL (10 seconds) lost its pins: `wavehouse_ingest_shards_unowned` counted its shards, and a peer that judged it dead reset them, redelivering rows it still held — two writers per table. Each shard now has a puller that never runs the handler, fetches only what the shard's cap leaves room for, and at the cap renews the pin every 5 seconds with a one-row fetch, at most one `ack_wait` of such rows past the cap, 12 at the generated 1 minute (past that, and once halted, with a pull of `max_bytes` 1, which delivers nothing); measured, an owner blocked 13 seconds on a hung insert kept one pin throughout and took 2 rows past its share, and a unit drains as fast as before (160,000–192,000 rows a second, against 158,000–172,000). `ResetOrphaned` and `Unowned` also require that the shard delivered nothing and had no ack for its pinned TTL. The process-wide cap of 10,000 held rows let one stuck table take every slot and stall every shard the process owned; each shard now holds its own share, plus at most the 12 rows its renewals take (see the entry above), and with one shard's share full a healthy shard's rows kept flowing. A clean stop now halts every shard first (`mq.Halter`), keeping the pins, so the worker writes what it holds and what the shards still deliver, and releases them after its final flush; rows delivered during that flush were dropped before, and came back only once they had been waited out. A handover keeps the pin until the rows delivered are written, bounded by the worker's 60-second ack wait instead of 15 seconds, and a row the worker drops while stopping is NAKed before the release. Checked under continuous publishing, with the first process's ClickHouse slower than a batch window: every table's rows arrived exactly once and in order across a handover and a clean stop, and the next owner wrote its first rows about 2 seconds after the stop, against 9 seconds when the stop leaves rows behind. A shard fetches in 1-second pulls, so a shard stops fetching within a second of a halt, and an idle shard costs one pull a second. A shard durable deleted while no pull of it was waiting stalled silently until the five-minute topology check, since the next pull only got no responders: two such pulls in a row, or one renewal, now look the durable up, and one that is gone, or whose stream is, ends the worker (measured: 5 seconds after the delete, with the handler busy). A bind whose durable or stream is gone, or whose durable no longer fits (`mq.ErrConsumerMismatch`: `ack_wait` shorter than asked, or no `max_ack_pending`), ends the worker instead of retrying every tick with a warning, and a bind failure that lasts a minute logs an error. `wavehouse_ingest_rows_held_waits_total` is gone: nothing waits at the cap any more. A renewal at the cap takes a row rather than holding it back: a held-back redelivery goes to the back of the server's redelivery queue, and with max_bytes-1 renewals rows NAKed as 0, 1, 2, 3 came back as 2, 3, 0, 1 (measured); now in order, until a stuck shard is 12 rows past its cap. The server requeues the same way while another process's pull waits for the pin, so rows redelivered across a handover or a stop may come back out of order among themselves, though still ahead of newer rows. The unpin on release names no pin id, so it is preceded by a check that the pin is still this process's; the unit renews its pin until the unpin returns (for up to `ack_wait` and one 5-second renewal after its halt has handed on what it fetched, longer than a handover waits), so the pin cannot lapse between the two, and only a consumer leader change there could move it. - **External NATS: the shipped manifests fit the shipped volume, boot refuses permissions narrower than the shards, and the tooling covers `js_domain`** (`internal/mq/{nats_manifests,nats_topology,external}.go`, `internal/mq/nats_permprobe.go` (new), `internal/mq/natstest/natstest.go`, `cmd/wavehouse/mq.go` (+ tests; `internal/mq/{nats_manifests,external_perms}_test.go` new), `deployments/nats/{jetstream.yaml,values.yaml}`, `docs/src/content/docs/{deployment,architecture}.md`, `configuration.mdx`): part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). The shipped streams reserved 225 GiB against the shipped 100 Gi volume, whose size the NATS Helm chart makes each server's `max_file_store`, so the server refused the third partition. `wavehouse mq manifests --file-store` (default `100Gi`) now sizes each partition at 15% of it, the history at 10% and the dead-letter stream at 5%, so the shipped four partitions reserve 75 GiB; a partition's size does not follow N, and the generator refuses streams that reserve more than the store (`--partition-max-bytes` resizes them). The test server takes its file store from the shipped values as the chart does, where it had a limit no test could reach. Boot now probes every shard durable as the connecting user, with a pull and an unpin the server rejects on their merits, and a `required` finding names each durable the user may not pull, where permissions generated for fewer shards passed boot and left those shards' rows unpulled; a consumer request the server refuses later sets `wavehouse_mq_topology_ok` to `0` at once. `wavehouse mq permissions --js-domain` allows and denies the `$JS..API` subjects a client in a JetStream domain sends, beside the plain ones a server in the domain checks after mapping them; `wavehouse mq manifests` takes `--ingest-consumer`, `--history-stream` and `--publish-timeout`, the duplicate window following the timeout; the `wavehouse` user may ack only for its shard durables, under either ack subject layout, where it could ack for any consumer; and an `mq.nats.ingest_consumer` or `history_stream` outside `[a-zA-Z0-9_-]` refuses boot. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index dd27e108..d3ab02ba 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -537,12 +537,12 @@ Every column the query references — in `columns`, an aggregation argument, `fi | `group_by` | string[] | No | GROUP BY columns. | | `order_by` | object[] | No | ORDER BY clauses (`column`, `dir`). | | `limit` | int | No | Max rows. Omitted or above the configured `query.default_max_rows` (default 10,000) → silently capped at that value; a policy `max_rows` can lower it further (see [Access Control](/access-control#resource-limits)). | -| `time_range` | object | No | Time window (`column`, `since`, `until`). `since`/`until` accept RFC3339 or Go-duration relative values ("1h", "30m", "7d", "2w" — day and week suffixes expand to hours). Relative values mean that long *ago*. The window applies only when `column` and `since` are set — an `until` without `since` is ignored. | +| `time_range` | object | No | Time window (`column`, `since`, `until`). `since`/`until` accept RFC3339 or Go-duration relative values ("1h", "30m", "7d", "2w" — day and week suffixes expand to hours). Relative values mean that long *ago*. Both bounds are instants: they reach ClickHouse as RFC 3339 in UTC, so the column's zone cannot shift them. The window applies only when `column` and `since` are set — an `until` without `since` is ignored. | :::note[Filter values bind as strings] -Every bound value — a caller's filter, a policy row filter, an insert `check` — binds as a `{p:String}` parameter and is compared under the column's own type. One rule across all three surfaces, and the answer is the server's: send ClickHouse's own spelling for a value and it reads it. A policy claim on an integer column is compared through a strict cast on top, so a claim that is not the canonical spelling of a value the column can hold matches nothing instead of wrapping (see [Access Control](/access-control#jwt-claim-templating)); a caller's own filter keeps the plain form, since it can only narrow what the policy admits. There is no RFC 3339 sniff any more: a value for a `DateTime` column is handed to ClickHouse verbatim, so a `"2026-01-01T00:00:00Z"` filter works exactly as far as the connected server's version reads that spelling. +Every bound value — a caller's filter, a policy row filter, an insert `check` — binds as a `{p:String}` parameter and is compared under the column's own type. One rule across all three surfaces, and the answer is the server's: send ClickHouse's own spelling for a value and it reads it. A policy claim on an integer column is compared through a strict cast on top, so a claim that is not the canonical spelling of a value the column can hold matches nothing instead of wrapping (see [Access Control](/access-control#jwt-claim-templating)); a caller's own filter keeps the plain form, since it can only narrow what the policy admits. On **this endpoint**, a caller's filter on a `Date`, `DateTime` or `DateTime64` column (`Nullable` or `LowCardinality` too) is parsed by ClickHouse rather than compared as text, because compared directly ClickHouse refuses RFC 3339 on those columns. The value still reaches the server as you wrote it, inside `parseDateTime64BestEffort(…)` with the column's declared zone: an offset or `Z` is the exact instant, a zone-less `"2026-06-21 13:00:00"` is local time in the column's zone (else the server's), a Unix-seconds number works, and a fraction compares exactly, so `> "…04:00:00.5Z"` on a whole-second column excludes `04:00:00`. A `Date` column takes the date of that instant in the server's zone. A value ClickHouse cannot parse is `400 clickhouse.rejected`. `like` compares text and is never parsed. A policy row filter keeps the plain form, so the query path and the stream judge it alike. -On **this endpoint** an `in` list binds as one `Array(String)` parameter, and ClickHouse caps a query-string parameter at about 64 KiB of literal text — roughly 6,000 short elements. Past that the query fails with ClickHouse's own `500` rather than a clean `400`; the 1 MiB request-body cap alone would have allowed more. Stream row filters bind their `in` elements one parameter each and carry no such ceiling. +An `in` list travels as a ClickHouse **external table**, not a query parameter, so no ClickHouse field limit applies to it: any list the 1 MiB request body can carry reaches the server. Each element is converted to the column's type (`accurateCastOrNull`, or the timestamp parse above), so `["12.5"]` matches a `Decimal` 12.50, and an element that is not a value of the column's type (`"256"` on a `UInt8`, `"1.5"` on an integer column) matches no row. Scalar values ride on the request line, where ClickHouse takes at most 128 KiB per value once URL-encoded and about 1 MiB for the whole line. With an `in` list the SQL itself travels in one 128 KiB form field. Past any of these the request is a `400` that names the limit, before anything is sent. ::: :::note[Identifier names] @@ -564,9 +564,9 @@ JSON array of result rows, **rendered by ClickHouse**: the query runs over its H | `Array`, `Map`, `Tuple` | ClickHouse's own JSON for the container | | `NaN` / `Inf` | `null` (ClickHouse's default rendering) | -Keys come back in **SELECT order**, not alphabetical. The query runs with `readonly=2` and a server-side `max_execution_time`, so a statement that writes cannot slip through a read path and a runaway query is stopped by ClickHouse rather than only abandoned by WaveHouse. A filter value of `null` is a `400`, and an `in` list travels as one `Array(String)` parameter with a size check that answers `400` before the request is sent. The response carries `X-Cache: HIT` or `X-Cache: MISS` — this endpoint shares the query cache + singleflight machinery (unlike `/v1/ops/query`, which always hits ClickHouse), keyed by [tenant](/deployment#multi-tenant-deployments): a request is never served from, or coalesced with, another tenant's. The cache stores ClickHouse's own bytes. +Keys come back in **SELECT order**, not alphabetical. The query runs with `readonly=2` and a server-side `max_execution_time`, so a statement that writes cannot slip through a read path and a runaway query is stopped by ClickHouse rather than only abandoned by WaveHouse. A filter value of `null` is a `400`, and an `in` list travels as an external table with no size cap of its own. The response carries `X-Cache: HIT` or `X-Cache: MISS` — this endpoint shares the query cache + singleflight machinery (unlike `/v1/ops/query`, which always hits ClickHouse), keyed by [tenant](/deployment#multi-tenant-deployments): a request is never served from, or coalesced with, another tenant's. The cache stores ClickHouse's own bytes. -The inbound request body is capped at 1 MiB; a body over the cap is rejected with `413`. A query AST is bounded by nature (far under 1 MiB even with a large `in`-list), and the cap blocks a single-request memory-exhaustion vector on this public endpoint. Set a tighter or higher outer limit at your [reverse proxy](/reverse-proxy#request-body-size-limits) — but it can only narrow the effective limit, not raise it past this cap. +The inbound request body is capped at 1 MiB; a body over the cap is rejected with `413`. It is also the only bound on an `in` list's length, and it blocks a single-request memory-exhaustion vector on this public endpoint. Set a tighter or higher outer limit at your [reverse proxy](/reverse-proxy#request-body-size-limits) — but it can only narrow the effective limit, not raise it past this cap. **Error responses:** @@ -574,6 +574,7 @@ The inbound request body is capped at 1 MiB; a body over the cap is rejected wit | ------ | ---- | ----- | | 400 | `{"error":"unknown column: x"}` | Schema validation error — an unknown column, a bad aggregation, or an unparseable `time_range` `since`/`until` (neither a relative duration nor an RFC3339 timestamp) | | 400 | `{"error":"filter value must not be null"}` | A filter carries `"value": null`. It is refused rather than answered: `col = NULL` is never true, and an empty parameter would silently ask a different question | +| 400 | `{"error":"filter value too large: …"}` / `{"error":"query too large: …"}` | A scalar filter value over the 128 KiB ClickHouse's HTTP interface takes for one value once URL-encoded, all of them over the request line, or a query with an `in` list whose SQL is over the 128 KiB form field it travels in. An `in` list itself is never refused for size | | 500 | `{"error":"clickhouse query: Code: 158. DB::Exception: … (TOO_MANY_ROWS) …"}` | ClickHouse refused the query — a resource limit, a type mismatch, anything else its engine raises. The body carries ClickHouse's own wording | | 403 | `{"error":"forbidden"}` | Role lacks select permission on table | | 403 | `{"error":"column \"x\" not allowed"}` | Column denied by policy | diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 7945b050..b8d1fb67 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -86,7 +86,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. - **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch, and the dedupe id is read positionally out of the exported row. One `typelayer.Table.Ingest` call per body, on the table set bound for the request's tenant, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5`, decided before the body is read and ahead of every other tenant's traffic. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent or `null` can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` — a deliberately different audience from the structured-query path, which leaves ClickHouse's default spelling alone so it matches the SSE wire. -- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every filter value bound as a named `{pN:String}` parameter, so ClickHouse renders each row and WaveHouse only frames the lines into an array. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=simple`) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. Connections per tenant are capped at the tenant's `max_open_conns`. +- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=simple`) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. Connections per tenant are capped at the tenant's `max_open_conns`. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) driver.Conn` beside it — each resolves the request's tenant per call, and a nil connection (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for one tenant's per-table parked counts (optionally one table) and its total — the tenant `?tenant=` names, read strictly by `opsTenant`, tenant `0` without it. The tenant is looked up in the MQ, not the settings registry, so a rejected or removed tenant's parked rows are read like a served one's; a tenant with no dead-letter queue (`mq.ErrNoDeadLetterQueue`) is a 404, and any other failure to read it a 500. The queue itself is `internal/mq`'s. @@ -223,6 +223,7 @@ Predicate *evaluation* lives elsewhere: `internal/typelayer` compiles a role's r ### `query/` — Structured Query Engine - **ast.go** — `StructuredQuery` AST types: columns, aggregations, filters, group by, order by, limit, time range. +- **bind.go** — `BuildResult.Bind()` turns the positional placeholders into what ClickHouse's HTTP interface takes: a `{pN:String}` parameter per scalar, wrapped in `parseDateTime64BestEffort` (with the column's declared zone; `toDate`/`toDate32` of it for a date) on a `Date`/`DateTime` column so ClickHouse reads the caller's own spelling, and an external table per `in` list, read by `IN (SELECT … FROM _pN)` with each element converted to the column's type. - **builder.go** — `Build()` converts AST to parameterized SQL. It is the single chokepoint that validates every referenced identifier against the schema **and** authorizes every column reference — projection, aggregation args, filters, group_by, order_by, time_range — against the role's column allowlist (the [#223](https://github.com/Wave-RF/WaveHouse/issues/223) hard cap). A full-row read is requested with `select_all`, which expands to the role's allowed columns rather than emitting a raw `SELECT *`; an omitted projection selects nothing, and `*` in `columns` is a literal column name. Every identifier is backtick-quoted via `internal/chsql` (`QuoteIdent`) so any ClickHouse-legal name is accepted — a name containing `?` is refused fail-closed, because `Build` emits positional placeholders that a later pass rewrites to ClickHouse's named parameters ([#279](https://github.com/Wave-RF/WaveHouse/issues/279)). The role's row-level-security predicate and `max_rows` cap are emitted by `Build()` itself, as part of the WHERE and LIMIT assembly — policy SQL is never spliced into rendered text ([#322](https://github.com/Wave-RF/WaveHouse/issues/322)). Timestamp bucketing for cache optimization. ### `settings/` — Settings Directory diff --git a/tests/e2e/sdk/query.test.ts b/tests/e2e/sdk/query.test.ts index 764871c4..fc336cb0 100644 --- a/tests/e2e/sdk/query.test.ts +++ b/tests/e2e/sdk/query.test.ts @@ -274,18 +274,19 @@ describe("Query", () => { } }); - // ClickHouse refuses an RFC 3339 String against a DateTime column, so the - // builder rewrites one on DateTime and DateTime64 columns; a timestamp read - // back from /v1/query, in ClickHouse's own spelling, filters as it is. + // ClickHouse parses a filter value on a DateTime column itself, so RFC 3339 + // with any offset is an exact instant, a zone-less value reads in the + // column's own zone, and a timestamp read back from /v1/query, in + // ClickHouse's own spelling, filters as it is. it("filters DateTime and DateTime64 columns by RFC 3339 and by the spelling it returns", async () => { const admin = adminClient(); const t = `ts_${testId().replace(/-/g, "_")}`; await chQuery( - `CREATE TABLE IF NOT EXISTS default.\`${t}\` (id String, at DateTime('UTC'), at3 DateTime64(3, 'UTC')) ENGINE = Memory`, + `CREATE TABLE IF NOT EXISTS default.\`${t}\` (id String, at DateTime('UTC'), at3 DateTime64(3, 'UTC'), atk DateTime('Asia/Tokyo')) ENGINE = Memory`, ); await chQuery( - `INSERT INTO default.\`${t}\` VALUES ('r1', '2026-01-15 10:30:00', '2026-01-15 10:30:00.123'), ('r2', '2026-01-16 00:00:00', '2026-01-16 00:00:00.000')`, + `INSERT INTO default.\`${t}\` VALUES ('r1', '2026-01-15 10:30:00', '2026-01-15 10:30:00.123', '2026-01-15 19:30:00'), ('r2', '2026-01-16 00:00:00', '2026-01-16 00:00:00.000', '2026-01-16 09:00:00')`, ); await admin.schema.refresh(); @@ -310,15 +311,33 @@ describe("Query", () => { }; expect(await ids("at", "=", "2026-01-15T10:30:00Z")).toEqual(["r1"]); expect(await ids("at", ">=", "2026-01-15T12:00:00+02:00")).toEqual(["r1", "r2"]); + expect(await ids("at", ">", "2026-01-15T10:30:00.5Z")).toEqual(["r2"]); expect(await ids("at3", "=", "2026-01-15T10:30:00.123Z")).toEqual(["r1"]); expect(await ids("at", "in", ["2026-01-16T00:00:00Z"])).toEqual(["r2"]); + // A Tokyo column: the same instant whatever the spelling, and a + // zone-less value in Tokyo's own time. + expect(await ids("atk", "=", "2026-01-15T10:30:00Z")).toEqual(["r1"]); + expect(await ids("atk", "=", "2026-01-15 19:30:00")).toEqual(["r1"]); + expect(await ids("atk", "in", ["2026-01-16T09:00:00+09:00"])).toEqual(["r2"]); - const back = await wh.from(t).select("id", "at", "at3").where("id", "=", "r1").fetch(); + const back = await wh.from(t).select("id", "at", "at3", "atk").where("id", "=", "r1").fetch(); expect(back.error).toBeNull(); - const row = back.data![0] as { at: string; at3: string }; + const row = back.data![0] as { at: string; at3: string; atk: string }; expect(row.at).toBe("2026-01-15 10:30:00"); + expect(row.atk).toBe("2026-01-15 19:30:00"); expect(await ids("at", "=", row.at)).toEqual(["r1"]); expect(await ids("at3", "=", row.at3)).toEqual(["r1"]); + expect(await ids("atk", "=", row.atk)).toEqual(["r1"]); + + // An in list far past the 128 KiB a ClickHouse query parameter takes + // still reaches ClickHouse whole; only the 1 MiB request body bounds it. + const many = Array.from( + { length: 20_000 }, + (_, i) => `2027-01-01T00:00:${String(i % 60).padStart(2, "0")}.${i}Z`, + ); + expect(await ids("at3", "in", [...many, "2026-01-15T10:30:00.123Z"])).toEqual(["r1"]); + const names = Array.from({ length: 30_000 }, (_, i) => `no-such-id-${i}`); + expect(await ids("id", "in", [...names, "r2"])).toEqual(["r2"]); } finally { await setPolicy(currentPolicy); await chQuery(`DROP TABLE IF EXISTS default.\`${t}\``); From 39d3528aa7e1f56cfda0183ddda3904b455d192d Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:46:14 -0400 Subject: [PATCH 23/70] docs(api): give the timestamp-filter rule its own paragraph Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- docs/src/content/docs/api.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 94cfe175..5908ff9d 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -542,7 +542,9 @@ Every column the query references — in `columns`, an aggregation argument, `fi | `time_range` | object | No | Time window (`column`, `since`, `until`). `since`/`until` accept RFC3339 or Go-duration relative values ("1h", "30m", "7d", "2w" — day and week suffixes expand to hours). Relative values mean that long *ago*. Both bounds are instants: they reach ClickHouse as RFC 3339 in UTC, so the column's zone cannot shift them. The window applies only when `column` and `since` are set — an `until` without `since` is ignored. | :::note[Filter values bind as strings] -Every bound value — a caller's filter, a policy row filter, an insert `check` — binds as a `{p:String}` parameter and is compared under the column's own type. One rule across all three surfaces, and the answer is the server's: send ClickHouse's own spelling for a value and it reads it. A policy claim on an integer column is compared through a strict cast on top, so a claim that is not the canonical spelling of a value the column can hold matches nothing instead of wrapping (see [Access Control](/access-control#jwt-claim-templating)); a caller's own filter keeps the plain form, since it can only narrow what the policy admits. On **this endpoint**, a caller's filter on a `Date`, `DateTime` or `DateTime64` column (`Nullable` or `LowCardinality` too) is parsed by ClickHouse rather than compared as text, because compared directly ClickHouse refuses RFC 3339 on those columns. The value still reaches the server as you wrote it, inside `parseDateTime64BestEffort(…)` with the column's declared zone: an offset or `Z` is the exact instant, a zone-less `"2026-06-21 13:00:00"` is local time in the column's zone (else the server's), a Unix-seconds number works, and a fraction compares exactly, so `> "…04:00:00.5Z"` on a whole-second column excludes `04:00:00`. A `Date` column takes the date of that instant in the server's zone. A value ClickHouse cannot parse is `400 clickhouse.rejected`. `like` compares text and is never parsed. A policy row filter keeps the plain form, so the query path and the stream judge it alike. +Every bound value — a caller's filter, a policy row filter, an insert `check` — binds as a `{p:String}` parameter and is compared under the column's own type. One rule across all three surfaces, and the answer is the server's: send ClickHouse's own spelling for a value and it reads it. A policy claim on an integer column is compared through a strict cast on top, so a claim that is not the canonical spelling of a value the column can hold matches nothing instead of wrapping (see [Access Control](/access-control#jwt-claim-templating)); a caller's own filter keeps the plain form, since it can only narrow what the policy admits. + +On **this endpoint**, a caller's filter on a `Date`, `DateTime` or `DateTime64` column (`Nullable` or `LowCardinality` too) is parsed by ClickHouse rather than compared as text, because compared directly ClickHouse refuses RFC 3339 on those columns. The value still reaches the server as you wrote it, inside `parseDateTime64BestEffort(…)` with the column's declared zone: an offset or `Z` is the exact instant, a zone-less `"2026-06-21 13:00:00"` is local time in the column's zone (else the server's), a Unix-seconds number works, and a fraction compares exactly, so `> "…04:00:00.5Z"` on a whole-second column excludes `04:00:00`. A `Date` column takes the date of that instant in the server's zone. A value ClickHouse cannot parse is `400 clickhouse.rejected`. `like` compares text and is never parsed. A policy row filter keeps the plain form, so the query path and the stream judge it alike. An `in` list travels as a ClickHouse **external table**, not a query parameter, so no ClickHouse field limit applies to it: any list the 1 MiB request body can carry reaches the server. Each element is converted to the column's type (`accurateCastOrNull`, or the timestamp parse above), so `["12.5"]` matches a `Decimal` 12.50, and an element that is not a value of the column's type (`"256"` on a `UInt8`, `"1.5"` on an integer column) matches no row. Scalar values ride on the request line, where ClickHouse takes at most 128 KiB per value once URL-encoded and about 1 MiB for the whole line. With an `in` list the SQL itself travels in one 128 KiB form field. Past any of these the request is a `400` that names the limit, before anything is sent. ::: From 943a82814f35afe70732726d838ca3ab978added Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:50:00 -0400 Subject: [PATCH 24/70] docs(access-control): timestamp claims take the zone-less form ClickHouse refuses an RFC 3339 string compared with a DateTime column (code 53) on 24.8 and 26.8 alike, so the page no longer promises offsets from 26.5 on. The chsql note about Array(String) quoting goes with the code it described. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- docs/src/content/docs/access-control.mdx | 2 +- internal/chsql/chsql.go | 6 ------ 2 files changed, 1 insertion(+), 7 deletions(-) diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index a04d0127..581447ba 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -252,7 +252,7 @@ Compare claims against `String` or `UUID` columns (tenant ids, org ids, user ids On an integer column (`UInt8` through `UInt256`, `Int8` through `Int256`, `Nullable` included) the claim must still fit the column's type, and is compared through a strict cast: a claim that is not the canonical spelling of a value the column can hold — out of range such as `18446744073709551616` (2^64), or spelled `007` or `+5` — matches no rows on any operator, and an insert `check` refuses the record with `403`. It never wraps onto another value. -On a timestamp column (`Date`, `DateTime`, `DateTime64`) the claim is parsed the way ClickHouse parses a string compared with that type. A claim without a UTC offset is read in the server's time zone, and one carrying an offset such as `2026-01-01T00:00:00Z` is accepted only on ClickHouse 26.5 and later — earlier lines refuse the comparison. +On a timestamp column (`Date`, `DateTime`, `DateTime64`) the claim is compared the way ClickHouse compares a string with that type. Write it as `YYYY-MM-DD hh:mm:ss[.fff]` with no offset: it is read in the column's time zone, or the server's when the column declares none. A claim in RFC 3339 form such as `2026-01-01T00:00:00Z` is refused on a `DateTime` column (ClickHouse code 53, measured on 24.8 and 26.8) rather than matched, so for timestamps use the zone-less form, or compare claims against a String or UUID column instead. Prefer identity columns over time columns, and express time windows in the query instead. ::: diff --git a/internal/chsql/chsql.go b/internal/chsql/chsql.go index 47d2cb47..8c7f6ec9 100644 --- a/internal/chsql/chsql.go +++ b/internal/chsql/chsql.go @@ -67,12 +67,6 @@ func BindUnsafe(name string) bool { // compare equal. Encoding `\` → `\\`, tab → `\t`, newline → `\n`, CR → `\r` // round-trips every value byte for byte on both surfaces, including an // embedded NUL. -// -// An Array(String) parameter takes a DIFFERENT rule and must NOT be run -// through this one: its elements are read as QUOTED values, where a raw tab or -// newline rides through untouched and only `'` and `\` need escaping. -// Applying both encodings is corruption — `a\b` becomes `a\\b`. See -// quoteCHElement in internal/query. var EscapeStringParam = strings.NewReplacer( `\`, `\\`, "\t", `\t`, From 9513fdba652cf8faeacf712830aa93fa553b8dd9 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 06:59:49 -0400 Subject: [PATCH 25/70] test(integration): fit the suite back under its timeout The integration package hit its 480s timeout after the chtypes work. Per test, against main (same flags, same Mac): the stream/query differential added 54s, the new ingest tests about 63s (each waits out the worker's 5s batch window), and the three tests that stop ClickHouse took 33s longer, because 26.8 outlives their 10s stop timeout. - Differential: the stream side prepares each stored row once and asks every cell of it (the hub's own shape: one parse per event, a Visible per subscriber); the 8,838 /v1/query requests go 8 at a time over one keep-alive pool with tokens minted once per role. Every cell is still compared through the production handler; a non-200 other than 400 clickhouse.rejected still fails the test, as does no admitted cell. - Ingest tests that wait out the batch window run in parallel. withPolicy now adopts the union of the grants every running test holds, so tests with policies on their own tables can overlap; createTable serializes its create-and-refresh, since two overlapping refreshes publish in the order they finish. - Test ClickHouse containers cap shutdown_wait_unfinished at 1s. On SIGTERM 26.8 waits for idle client connections (a native one until its 10s poll interval, an HTTP keep-alive one until its 30s timeout), so a stop with one idle native connection took 13s (5s on 26.6), 1s with the setting. - The settings directory gets a scratch parent of its own: the watcher watches the parent, and under kqueue watching the shared temp directory holds and rescans every entry in it. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- tests/integration/coord_nats_test.go | 2 + tests/integration/ingest_test.go | 18 ++ tests/integration/query_errors_test.go | 3 +- tests/integration/rowfilter_stream_test.go | 301 ++++++++++++++------- tests/integration/setup_test.go | 123 +++++++-- tests/integration/shared_cache_test.go | 1 + tests/integration/typelayer_wire_test.go | 1 + 7 files changed, 327 insertions(+), 122 deletions(-) diff --git a/tests/integration/coord_nats_test.go b/tests/integration/coord_nats_test.go index 0ca60ed8..5715ef9c 100644 --- a/tests/integration/coord_nats_test.go +++ b/tests/integration/coord_nats_test.go @@ -33,6 +33,7 @@ func TestCoordNATS_ReplicasShareTheShards(t *testing.T) { require.NoError(t, op.ApplyShipped(ctx)) root, err := writeTestSettings(e.ch) require.NoError(t, err) + t.Cleanup(func() { removeTestSettings(root) }) replicas := map[string]*natsProcess{} for range 2 { @@ -83,6 +84,7 @@ func TestCoordNATS_MissingBucketRefusesBoot(t *testing.T) { require.NoError(t, os.WriteFile(pw, []byte(natstest.Password(natstest.WaveHouseUser)), 0o600)) root, err := writeTestSettings(env(t).ch) require.NoError(t, err) + t.Cleanup(func() { removeTestSettings(root) }) cfg := &config.Config{ DataDir: t.TempDir(), Server: config.Server{Port: 1, ShutdownTimeout: 1}, diff --git a/tests/integration/ingest_test.go b/tests/integration/ingest_test.go index 8b5cee95..21ecf199 100644 --- a/tests/integration/ingest_test.go +++ b/tests/integration/ingest_test.go @@ -163,6 +163,10 @@ func postIngest(t *testing.T, table, contentType, body, authorization string) (i // rows matching where to be want. A want of 0 holds only once something posted // later has landed, so the tests asserting an absence post a record that // lands after the refused ones and wait for it first. +// +// Each wait is most of the worker's batch window, idle, so the tests here +// that wait run in parallel: each writes and reads a table of its own, and a +// policy it adopts grants on that table alone (withPolicy holds the union). func eventuallyRows(t *testing.T, table, where string, want uint64) { t.Helper() ctx := context.Background() @@ -181,6 +185,7 @@ func eventuallyRows(t *testing.T, table, where string, want uint64) { // newlines in place, which restores per-record salvage. Asserted in the table, // not only in the response. func TestIngest_CompactArray_OneBadRecord_TheOthersStillLand(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, value UInt32", "ORDER BY user_id") status, body := postIngest(t, table, "application/json", @@ -198,6 +203,7 @@ func TestIngest_CompactArray_OneBadRecord_TheOthersStillLand(t *testing.T) { // assertion is what makes the positional contract real: a column-order bug // would still answer 200. func TestIngest_CSVBody_LandsInClickHouse(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") status, body := postIngest(t, table, "text/csv", "\"c1\",\"click\",7\n\"c2\",\"view\",9\n", "") @@ -211,6 +217,7 @@ func TestIngest_CSVBody_LandsInClickHouse(t *testing.T) { // A bare text/csv is ClickHouse's default CSV, so a first line spelling the // column names is consumed as a header and only the data rows land. func TestIngest_BareCSVHeader_LandsInClickHouse(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") status, body := postIngest(t, table, "text/csv", "user_id,event_type,value\n\"d1\",\"click\",7\n", "") @@ -225,6 +232,7 @@ func TestIngest_BareCSVHeader_LandsInClickHouse(t *testing.T) { // header=absent is strictly positional, so the same header line is one // refused record and never reaches the table. func TestIngest_CSVHeaderAbsent_LandsInClickHouse(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") status, body := postIngest(t, table, "text/csv; header=absent", "user_id,event_type,value\n\"a1\",\"click\",7\n", "") @@ -240,6 +248,7 @@ func TestIngest_CSVHeaderAbsent_LandsInClickHouse(t *testing.T) { // TSV is CSV's tab-separated twin, with a bad row beside a good one, so // per-record salvage is covered for the positional formats too. func TestIngest_TSVBody_LandsInClickHouse(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") status, body := postIngest(t, table, "text/tab-separated-values", "t1\tclick\t5\nt2\tview\tnope\n", "") @@ -255,6 +264,7 @@ func TestIngest_TSVBody_LandsInClickHouse(t *testing.T) { // the header is not a record, a column it omits takes the table's own // DEFAULT, and a bad row is salvaged like any other. func TestIngest_CSVWithNamesBody_LandsInClickHouse(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, event_type String, value UInt32 DEFAULT 42", "ORDER BY user_id") status, body := postIngest(t, table, "text/csv; header=present", "event_type,user_id\nclick,h1\nview,h2\n", "") @@ -269,6 +279,7 @@ func TestIngest_CSVWithNamesBody_LandsInClickHouse(t *testing.T) { // The tab-separated twin of the header=present case, with a bad row beside a // good one. func TestIngest_TSVWithNamesBody_LandsInClickHouse(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, event_type String, value UInt32", "ORDER BY user_id") status, body := postIngest(t, table, "text/tab-separated-values; header=present", @@ -285,6 +296,7 @@ func TestIngest_TSVWithNamesBody_LandsInClickHouse(t *testing.T) { // refusal of the body, before any record: a whole-request 400 carrying its // code, and nothing stored. func TestIngest_WithNamesUnknownHeader_Is400WithCode117(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, value UInt32", "ORDER BY user_id") status, body := postIngest(t, table, "text/csv; header=present", "user_id,nosuch\nu1,1\n", "") @@ -305,6 +317,7 @@ func TestIngest_WithNamesUnknownHeader_Is400WithCode117(t *testing.T) { // writing without it still lands, the column taking the table's default // rather than any value of the caller's. func TestIngest_DeniedColumn_IsClickHouseCode117(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, secret String", "ORDER BY user_id") withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{ table: {"writer": {Insert: &policy.InsertPermissions{DenyColumns: []string{"secret"}}}}, @@ -326,6 +339,7 @@ func TestIngest_DeniedColumn_IsClickHouseCode117(t *testing.T) { // record's own value when it matches, and refuses one that does not (403, // nothing stored). func TestIngest_AutoInject_FillsAnAbsentCheckColumn(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, tenant String", "ORDER BY user_id") tmpl := "{{ jwt.tenant }}" withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{ @@ -350,6 +364,7 @@ func TestIngest_AutoInject_FillsAnAbsentCheckColumn(t *testing.T) { // is judged on the TABLE's default rather than refused for the absence. Both // directions, because the outcome is entirely what the default happens to be. func TestIngest_CheckIn_AbsentColumn_TestsTheTableDefault(t *testing.T) { + t.Parallel() tmpl := "{{ jwt.tenants }}" policyFor := func(table string) policy.Policy { return policy.Policy{Tables: map[string]policy.TablePolicy{ @@ -359,6 +374,7 @@ func TestIngest_CheckIn_AbsentColumn_TestsTheTableDefault(t *testing.T) { claims := map[string]any{"tenants": []any{"acme", "globex"}} t.Run("a default outside the set is refused", func(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, tenant String", "ORDER BY user_id") withPolicy(t, policyFor(table)) writer := bearer(t, "writer", claims) @@ -372,6 +388,7 @@ func TestIngest_CheckIn_AbsentColumn_TestsTheTableDefault(t *testing.T) { }) t.Run("a default inside the set is admitted", func(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, tenant String DEFAULT 'acme'", "ORDER BY user_id") withPolicy(t, policyFor(table)) status, body := postIngest(t, table, "application/json", `{"user_id":"n3"}`, bearer(t, "writer", claims)) @@ -388,6 +405,7 @@ func TestIngest_CheckIn_AbsentColumn_TestsTheTableDefault(t *testing.T) { // supplies the wrapped value itself; the same policy with a claim that fits // lands as before. func TestIngest_IntegerCheckClaimThatDoesNotFit_IsRefused(t *testing.T) { + t.Parallel() table := createTable(t, "user_id String, tenant UInt64", "ORDER BY user_id") tmpl := "{{ jwt.tenant }}" withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{ diff --git a/tests/integration/query_errors_test.go b/tests/integration/query_errors_test.go index fed6ecbc..276f28d6 100644 --- a/tests/integration/query_errors_test.go +++ b/tests/integration/query_errors_test.go @@ -10,7 +10,6 @@ import ( "net" "net/http" "net/url" - "os" "strings" "testing" "time" @@ -121,7 +120,7 @@ func TestQueryErrors_ClickHouseDown(t *testing.T) { settingsDir, err := writeTestSettings(ch) require.NoError(t, err) - t.Cleanup(func() { _ = os.RemoveAll(settingsDir) }) + t.Cleanup(func() { removeTestSettings(settingsDir) }) var lc net.ListenConfig ln, err := lc.Listen(ctx, "tcp", "127.0.0.1:0") diff --git a/tests/integration/rowfilter_stream_test.go b/tests/integration/rowfilter_stream_test.go index e913a735..c77acc99 100644 --- a/tests/integration/rowfilter_stream_test.go +++ b/tests/integration/rowfilter_stream_test.go @@ -13,6 +13,8 @@ import ( "net/url" "strconv" "strings" + "sync" + "sync/atomic" "testing" "github.com/stretchr/testify/require" @@ -173,22 +175,11 @@ func TestRowFilterStream_DifferentialAgainstClickHouse(t *testing.T) { }, } - // One row under test per (shape, payload): the table it lives in, its id, and - // the positional line the ingest path would publish for it. - type storedRow struct { - shape string - intType string - table string - id uint32 - payload any - columns []string - line []byte - } - var rows []storedRow - - // One policy for the whole corpus: on each shape's table, a role per - // (constant, operator) cell filtering v by that one predicate, and on an - // integer table one more whose _in reads a claim array. + // One row under test per (shape, payload), and one policy for the whole + // corpus: on each shape's table, a role per (constant, operator) cell + // filtering v by that one predicate, and on an integer table one more whose + // _in reads a claim array. + var rows []diffRow tables := map[string]policy.TablePolicy{} for _, sh := range shapes { table := createTable(t, "id UInt32, v "+sh.ddl, "ORDER BY id") @@ -213,10 +204,10 @@ func TestRowFilterStream_DifferentialAgainstClickHouse(t *testing.T) { continue } anyStored = true - rows = append(rows, storedRow{ + line := rowFilterStoredLine(t, table, uint32(i)) + rows = append(rows, diffRow{ shape: sh.name, intType: sh.intType, table: table, id: uint32(i), payload: payload, - columns: []string{"id", "v"}, - line: rowFilterStoredLine(t, table, uint32(i)), + columns: []string{"id", "v"}, line: line, stored: rowFilterStoredInt(t, sh.intType, line), }) } require.True(t, anyStored, "corpus for %s must contain insertable payloads", sh.name) @@ -224,67 +215,96 @@ func TestRowFilterStream_DifferentialAgainstClickHouse(t *testing.T) { p := policy.Policy{Tables: tables} withPolicy(t, p) - // The hub's evaluator over the app's type layer, which createTable's - // refreshes already bound. - require.NotNil(t, env(t).types, "the api role wires a type layer") - eval := stream.NewRowEvaluator(env(t).types) - + // Every cell, row by row: each of the shape's (constant, operator) roles, and + // on an integer column each claim array under the _in role. byShape := map[string][]string{} for _, sh := range shapes { byShape[sh.name] = sh.constants } - cells, mathCells, admitted := 0, 0, 0 - for _, r := range rows { - stored := rowFilterStoredInt(t, r.intType, r.line) + var cells []diffCell + for ri, r := range rows { for i, constant := range byShape[r.shape] { for _, op := range diffOps { - cells++ - role := cellRole(i, op) - got := rowFilterStreamVerdict(t, eval, &p, role, r.table, r.columns, r.line, nil) - want, sqlErr := rowFilterQueryVerdict(t, r.table, r.id, role, nil) - if want { - admitted++ - } - if got != want { - t.Errorf("%s: stored %v %s %q — stream says %v, /v1/query says %v (query err: %v)", - r.shape, r.payload, op.sql, constant, got, want, sqlErr) - } - if exact, ok := intExpected(r.intType, stored, op.sql, constant); ok { - mathCells++ - if want != exact || got != exact { - t.Errorf("%s: stored %v %s %q — admitted stream=%v query=%v, the mathematically correct answer is %v", - r.shape, r.payload, op.sql, constant, got, want, exact) - } - } + cells = append(cells, diffCell{row: ri, role: cellRole(i, op), op: op.sql, constant: constant}) } } - - // A multi-element _in from a claim array: each element is cast on its - // own, so the elements that fit decide and the rest drop out — 2^64+5 - // must not wrap onto a row holding 5. if r.intType != "" { for _, set := range intInSets { - cells++ - mathCells++ - claims := map[string]any{"ids": toAnys(set)} - got := rowFilterStreamVerdict(t, eval, &p, claimInRole, r.table, r.columns, r.line, claims) - want, sqlErr := rowFilterQueryVerdict(t, r.table, r.id, claimInRole, claims) - exact := false - for _, c := range set { - if eq, ok := intExpected(r.intType, stored, "=", c); ok && eq { - exact = true - } - } - if got != want || want != exact { - t.Errorf("%s: stored %v IN %v — stream says %v, /v1/query says %v (query err: %v), correct is %v", - r.shape, r.payload, set, got, want, sqlErr, exact) + cells = append(cells, diffCell{row: ri, role: claimInRole, claims: map[string]any{"ids": toAnys(set)}, set: set}) + } + } + } + + // The hub's evaluator over the app's type layer, which createTable's + // refreshes already bound. + require.NotNil(t, env(t).types, "the api role wires a type layer") + got := rowFilterStreamVerdicts(t, stream.NewRowEvaluator(env(t).types), &p, rows, cells) + want, sqlErrs := rowFilterQueryVerdicts(t, rows, cells) + + mathCells, admitted := 0, 0 + for i, c := range cells { + r := rows[c.row] + if c.set != nil { + // A multi-element _in from a claim array: each element is cast on its + // own, so the elements that fit decide and the rest drop out — 2^64+5 + // must not wrap onto a row holding 5. + mathCells++ + exact := false + for _, v := range c.set { + if eq, ok := intExpected(r.intType, r.stored, "=", v); ok && eq { + exact = true } } + if got[i] != want[i] || want[i] != exact { + t.Errorf("%s: stored %v IN %v — stream says %v, /v1/query says %v (query err: %v), correct is %v", + r.shape, r.payload, c.set, got[i], want[i], sqlErrs[i], exact) + } + continue + } + if want[i] { + admitted++ + } + if got[i] != want[i] { + t.Errorf("%s: stored %v %s %q — stream says %v, /v1/query says %v (query err: %v)", + r.shape, r.payload, c.op, c.constant, got[i], want[i], sqlErrs[i]) + } + if exact, ok := intExpected(r.intType, r.stored, c.op, c.constant); ok { + mathCells++ + if want[i] != exact || got[i] != exact { + t.Errorf("%s: stored %v %s %q — admitted stream=%v query=%v, the mathematically correct answer is %v", + r.shape, r.payload, c.op, c.constant, got[i], want[i], exact) + } } } // A harness that answered "no row" everywhere would agree with itself. require.Positive(t, admitted, "some cell must admit its row on the query path") - t.Logf("%d cells compared stream against /v1/query (%d admitted), %d of them also against the exact answer", cells, admitted, mathCells) + t.Logf("%d cells compared stream against /v1/query (%d admitted), %d of them also against the exact answer", len(cells), admitted, mathCells) +} + +// diffRow is one row under test: the table it lives in, its id, the +// positional line the ingest path would publish for it, and on an integer +// column the value stored (nil for NULL). +type diffRow struct { + shape string + intType string + table string + id uint32 + payload any + columns []string + line []byte + stored *big.Int +} + +// diffCell is one comparison: a stored row read as one role. A constant cell +// names its operator and constant; a claim-array cell runs as claimInRole with +// set as the ids claim. +type diffCell struct { + row int // index into the rows + role string + claims map[string]any + op string + constant string + set []string } // diffOp is one operator under the differential: its SQL spelling, for the @@ -326,46 +346,139 @@ func rowFilterGrant(t *testing.T, op diffOp, constant string) policy.RolePermiss return policy.RolePermissions{Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"v": f}}} } -// rowFilterStreamVerdict resolves role's grant through the full production -// path (Evaluate → Prepare → Visible) and reports whether the stream would -// deliver this stored row. -func rowFilterStreamVerdict(t *testing.T, eval stream.RowEvaluator, p *policy.Policy, role, table string, columns []string, line []byte, claims map[string]any) bool { +// rowFilterStreamVerdicts is the stream's verdict on each cell, reached the +// way the hub reaches it for an event: the row is prepared ONCE — the parse is +// per event — and each cell's grant, resolved through the full production path +// (Evaluate → Visible), is asked of that one view, as each subscriber's is. +// cells come row by row. +func rowFilterStreamVerdicts(t *testing.T, eval stream.RowEvaluator, p *policy.Policy, rows []diffRow, cells []diffCell) []bool { t.Helper() - perms := policy.Evaluate(p, role, table, "select", claims) - require.True(t, perms.Allowed) - - view, err := eval.Prepare(tenant.Default, table, columns, line) - if err != nil { - return false // withheld: no view, no row + got := make([]bool, len(cells)) + for start := 0; start < len(cells); { + r := rows[cells[start].row] + end := start + for end < len(cells) && cells[end].row == cells[start].row { + end++ + } + view, err := eval.Prepare(tenant.Default, r.table, r.columns, r.line) + for i := start; i < end; i++ { + perms := policy.Evaluate(p, cells[i].role, r.table, "select", cells[i].claims) + require.True(t, perms.Allowed) + if err == nil { // withheld otherwise: no view, no row + got[i], _ = view.Visible(perms) + } + } + if err == nil { + view.Close() + } + start = end } - defer view.Close() - visible, _ := view.Visible(perms) - return visible + return got } -// rowFilterQueryVerdict asks the production /v1/query, as role with claims, -// whether it returns the stored row. A query ClickHouse rejects means the role -// reads no rows on that path; any other failure is the harness's and fails the -// test, so a broken token or policy cannot pass as "withheld on both". -func rowFilterQueryVerdict(t *testing.T, table string, id uint32, role string, claims map[string]any) (bool, error) { +// diffQueryWorkers is how many /v1/query requests the differential keeps in +// flight: below the suite tenant's max_open_conns (10), so none of them waits +// on the server's own ClickHouse pool. One at a time, the round trips were +// most of a minute of the suite. +const diffQueryWorkers = 8 + +// rowFilterQueryVerdicts asks the production /v1/query, for every cell, whether +// it returns the stored row to the cell's role with the cell's claims, and +// returns each answer with ClickHouse's rejection when there was one. A query +// ClickHouse rejects means the role reads no rows on that path; any other +// failure is the harness's and fails the test, so a broken token or policy +// cannot pass as "withheld on both". The cells are independent reads, so they +// go diffQueryWorkers at a time over one pool of keep-alive connections. +func rowFilterQueryVerdicts(t *testing.T, rows []diffRow, cells []diffCell) ([]bool, []error) { t.Helper() + // Tokens are minted here, on the test's goroutine, one per role and claim set. + auth := make([]string, len(cells)) + minted := map[string]string{} + for i, c := range cells { + key := c.role + "\x00" + strings.Join(c.set, "\x00") + if _, ok := minted[key]; !ok { + minted[key] = bearer(t, c.role, c.claims) + } + auth[i] = minted[key] + } + + transport := http.DefaultTransport.(*http.Transport).Clone() + transport.MaxConnsPerHost, transport.MaxIdleConnsPerHost = diffQueryWorkers, diffQueryWorkers + defer transport.CloseIdleConnections() + client := &http.Client{Transport: transport} + + want := make([]bool, len(cells)) + sqlErrs := make([]error, len(cells)) + harnessErrs := make([]error, len(cells)) + var next atomic.Int64 + var broken atomic.Bool + var wg sync.WaitGroup + for range diffQueryWorkers { + wg.Go(func() { + for !broken.Load() { + i := int(next.Add(1) - 1) + if i >= len(cells) { + return + } + want[i], sqlErrs[i], harnessErrs[i] = rowFilterQueryVerdict(client, rows[cells[i].row], auth[i]) + if harnessErrs[i] != nil { + broken.Store(true) + } + } + }) + } + wg.Wait() + for i, err := range harnessErrs { + if err != nil { + t.Fatalf("role %s on %s: %v", cells[i].role, rows[cells[i].row].table, err) + } + } + return want, sqlErrs +} + +// rowFilterQueryVerdict is one cell's /v1/query: whether r comes back, the +// rejection when ClickHouse refused the query, or the harness failure. +func rowFilterQueryVerdict(client *http.Client, r diffRow, authorization string) (visible bool, rejected, harness error) { body, err := json.Marshal(map[string]any{ "columns": []string{"id"}, - "filters": []any{map[string]any{"column": "id", "op": "eq", "value": id}}, + "filters": []any{map[string]any{"column": "id", "op": "eq", "value": r.id}}, }) - require.NoError(t, err) - got := postJSONAs(t, env(t).baseURL+"/v1/query?table="+url.QueryEscape(table), string(body), bearer(t, role, claims)) - switch { - case got.status == http.StatusOK: - case got.status == http.StatusBadRequest && got.Code == "clickhouse.rejected": - return false, fmt.Errorf("HTTP %d: %s", got.status, got.raw) - default: - t.Fatalf("role %s on %s: HTTP %d: %s", role, table, got.status, got.raw) + if err != nil { + return false, nil, err + } + req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, + sharedEnv.baseURL+"/v1/query?table="+url.QueryEscape(r.table), bytes.NewReader(body)) + if err != nil { + return false, nil, err + } + req.Header.Set("Content-Type", "application/json") + req.Header.Set("Authorization", authorization) + resp, err := client.Do(req) + if err != nil { + return false, nil, err + } + defer func() { _ = resp.Body.Close() }() + raw, err := io.ReadAll(resp.Body) + if err != nil { + return false, nil, err + } + if resp.StatusCode == http.StatusBadRequest { + var refusal queryError + if json.Unmarshal(raw, &refusal) == nil && refusal.Code == "clickhouse.rejected" { + return false, fmt.Errorf("HTTP %d: %s", resp.StatusCode, raw), nil + } + } + if resp.StatusCode != http.StatusOK { + return false, nil, fmt.Errorf("HTTP %d: %s", resp.StatusCode, raw) } var out []map[string]any - require.NoError(t, json.Unmarshal([]byte(got.raw), &out)) - require.LessOrEqual(t, len(out), 1, "id is unique per table") - return len(out) == 1, nil + if err := json.Unmarshal(raw, &out); err != nil { + return false, nil, fmt.Errorf("decode %s: %w", raw, err) + } + if len(out) > 1 { + return false, nil, fmt.Errorf("id is unique per table, yet %d rows came back: %s", len(out), raw) + } + return len(out) == 1, nil, nil } // Boundary constants for the integer shapes: each width's own edges, the diff --git a/tests/integration/setup_test.go b/tests/integration/setup_test.go index dadef58b..54593d04 100644 --- a/tests/integration/setup_test.go +++ b/tests/integration/setup_test.go @@ -24,6 +24,7 @@ import ( "slices" "strconv" "strings" + "sync" "sync/atomic" "testing" "time" @@ -95,10 +96,18 @@ func env(t *testing.T) *testEnv { // (or sequentially-run) tests don't collide on table state. var tableCounter atomic.Uint64 +// createMu serializes createTable's create-and-refresh. +var createMu sync.Mutex + // createTable creates a uniquely-named ClickHouse table for the calling test // and registers cleanup to drop it. The schema registry is refreshed after // creation so the API discovers the new table. Returns the table name. // +// Parallel tests create tables concurrently, and two overlapping refreshes +// publish in the order they finish, not the order they read system.columns: +// an older snapshot landing last would drop the newer table from the +// registry. createMu makes each create-and-refresh one step. +// // Pass the column DDL fragment without the wrapping `()` — for example: // // createTable(t, "user_id String, value Float64", "ORDER BY user_id") @@ -114,6 +123,8 @@ func createTable(t *testing.T, columns, tableOpts string) string { "CREATE TABLE IF NOT EXISTS %s (%s) ENGINE = MergeTree() %s", name, columns, tableOpts, ) + createMu.Lock() + defer createMu.Unlock() if err := sharedEnv.chConn.Exec(ctx, stmt); err != nil { t.Fatalf("create test table %s: %v", name, err) } @@ -179,7 +190,7 @@ func setup() (int, func()) { fmt.Fprintf(os.Stderr, "integration setup: settings: %v\n", err) return 1, cleanup } - cleanups.push(func() { _ = os.RemoveAll(settingsDir) }) + cleanups.push(func() { removeTestSettings(settingsDir) }) // The wired app on a harness listener: the same construction the binary // uses (embedded NATS in-process, the ingest worker, sweeper, hub bridge, @@ -263,18 +274,27 @@ func setup() (int, func()) { // of its own (withPolicy) and sends a token for the role (bearer). The stream // budget is shrunk to 1 GiB like the e2e fixture so the scratch directory // stays small. +// +// The directory sits in a scratch parent of its own, removed with it +// (removeTestSettings): the settings watcher watches the parent too, and +// where fsnotify is kqueue (macOS) watching the shared temp directory opens +// and rescans every entry in it, on every change any process makes there. func writeTestSettings(ch *chInstance) (string, error) { files, err := tenantSettings(ch, testCHDatabase) if err != nil { return "", err } - dir := mustTempDir() + dir := filepath.Join(mustTempDir(), "settings") if err := writeSettingsFiles(dir, files); err != nil { return "", err } return dir, nil } +// removeTestSettings removes a directory writeTestSettings made, with its +// scratch parent. +func removeTestSettings(dir string) { _ = os.RemoveAll(filepath.Dir(dir)) } + // tenantSettings is one tenant's four files: the seed with the ClickHouse // block pointed at the testcontainer's database, and the dev-style policy. func tenantSettings(ch *chInstance, database string) (map[string][]byte, error) { @@ -308,40 +328,80 @@ func tenantSettings(ch *chInstance, database string) (map[string][]byte, error) return files, nil } -// withPolicy adopts p as the shared app's access-control policy for the -// calling test, the way an operator changes one: policies.json rewritten, -// with roles.json declaring every role it grants, then a reload through the -// ops route. The suite's default policy is restored when the test ends. +// withPolicy adds p's table grants to the shared app's access-control policy +// for the calling test, the way an operator changes one: policies.json +// rewritten, with roles.json declaring every role it grants, then a reload +// through the ops route. The grants come out again when the test ends. // default_role and admin_role stay the admin role, so unauthenticated // requests keep running as a privileged caller and a restricted role is -// reached with a token for it (bearer). Not for parallel tests: there is one -// shared policy. +// reached with a token for it (bearer). +// +// Only p.Tables is adopted, and its tables must be the test's own: the +// adopted policy is the union of the grants every running test holds, so +// parallel tests can each hold some without replacing another's. func withPolicy(t *testing.T, p policy.Policy) { t.Helper() - p.DefaultRole, p.AdminRole = "admin", "admin" - roles := map[string]bool{"admin": true} - for _, grants := range p.Tables { - for role := range grants { - roles[role] = true + policyMu.Lock() + defer policyMu.Unlock() + for holder, grants := range heldGrants { + for table := range p.Tables { + if _, taken := grants[table]; taken && holder != t { + t.Fatalf("withPolicy: %s already holds grants on %s", holder.Name(), table) + } } } - rolesDoc, err := json.Marshal(settings.RolesFile{Roles: slices.Sorted(maps.Keys(roles))}) - if err != nil { - t.Fatalf("roles.json: %v", err) - } - policyDoc, err := json.Marshal(p) - if err != nil { - t.Fatalf("policies.json: %v", err) - } - // The roles go in before the policy granting them and come out after it, - // so the directory watcher, which may reload between the two files, only - // ever sees a valid pair. - adoptSettings(t, settingsFile{settings.FileRoles, rolesDoc}, settingsFile{settings.FilePolicies, policyDoc}) + heldGrants[t] = p.Tables + adoptHeldGrants(t, true) t.Cleanup(func() { - adoptSettings(t, settingsFile{settings.FilePolicies, defaultPolicies}, settingsFile{settings.FileRoles, defaultRoles}) + policyMu.Lock() + defer policyMu.Unlock() + delete(heldGrants, t) + adoptHeldGrants(t, false) }) } +// heldGrants is each running test's withPolicy grants; policyMu serializes +// the adoptions of their union. +var ( + policyMu sync.Mutex + heldGrants = map[*testing.T]map[string]policy.TablePolicy{} +) + +// adoptHeldGrants adopts the union of heldGrants, or the suite's default +// policy when none is held. Under policyMu. A role goes in before the policy +// granting it and comes out after it (grow says which this is), so the +// directory watcher, which may reload between the two files, only ever sees +// a valid pair. +func adoptHeldGrants(t *testing.T, grow bool) { + t.Helper() + rolesDoc, policyDoc := defaultRoles, defaultPolicies + if len(heldGrants) > 0 { + p := policy.Policy{DefaultRole: "admin", AdminRole: "admin", Tables: map[string]policy.TablePolicy{}} + roles := map[string]bool{"admin": true} + for _, grants := range heldGrants { + for table, perms := range grants { + p.Tables[table] = perms + for role := range perms { + roles[role] = true + } + } + } + var err error + if rolesDoc, err = json.Marshal(settings.RolesFile{Roles: slices.Sorted(maps.Keys(roles))}); err != nil { + t.Fatalf("roles.json: %v", err) + } + if policyDoc, err = json.Marshal(p); err != nil { + t.Fatalf("policies.json: %v", err) + } + } + rolesFile, policyFile := settingsFile{settings.FileRoles, rolesDoc}, settingsFile{settings.FilePolicies, policyDoc} + if grow { + adoptSettings(t, rolesFile, policyFile) + } else { + adoptSettings(t, policyFile, rolesFile) + } +} + // settingsFile is one file of the settings directory and its new content. type settingsFile struct { name string @@ -469,6 +529,17 @@ func startClickHouse(ctx context.Context) (*chInstance, error) { Image: "clickhouse/clickhouse-server:26.8.15.10", ExposedPorts: []string{"9000/tcp", "8123/tcp"}, Env: map[string]string{"CLICKHOUSE_PASSWORD": testCHPassword}, + // The tests that stop ClickHouse mid-run stop it as an outage, and the + // server would otherwise outlast their stop timeout: on SIGTERM, 26.8 + // waits for its idle client connections to close, which takes a + // native one its 10 s poll interval and an HTTP keep-alive one its + // 30 s keep-alive timeout. Measured with one idle native connection, a + // stop took 13 s (5 s on 26.6) and with this setting 1 s. + Files: []testcontainers.ContainerFile{{ + Reader: strings.NewReader("1"), + ContainerFilePath: "/etc/clickhouse-server/config.d/test_shutdown.xml", + FileMode: 0o644, + }}, WaitingFor: wait.ForAll( wait.ForListeningPort("9000/tcp"), wait.ForHTTP("/ping").WithPort("8123/tcp").WithStatusCodeMatcher(func(status int) bool { diff --git a/tests/integration/shared_cache_test.go b/tests/integration/shared_cache_test.go index 6ace9a76..87003807 100644 --- a/tests/integration/shared_cache_test.go +++ b/tests/integration/shared_cache_test.go @@ -71,6 +71,7 @@ func bootRedisApp(t *testing.T, redisAddr, prefix string, timeout time.Duration) ctx := context.Background() settingsDir, err := writeTestSettings(e.ch) require.NoError(t, err) + t.Cleanup(func() { removeTestSettings(settingsDir) }) var lc net.ListenConfig ln, err := lc.Listen(ctx, "tcp", "127.0.0.1:0") require.NoError(t, err) diff --git a/tests/integration/typelayer_wire_test.go b/tests/integration/typelayer_wire_test.go index e95a2830..77bd6ff9 100644 --- a/tests/integration/typelayer_wire_test.go +++ b/tests/integration/typelayer_wire_test.go @@ -30,6 +30,7 @@ import ( // so one end-to-end identity check is the whole remaining claim. Deliberately // one test, not a corpus. func TestTypelayerWire_PublishedRowIsTheStoredRow(t *testing.T) { + t.Parallel() // a table of its own: see eventuallyRows e := env(t) ctx := context.Background() before := time.Now().UTC().Add(-time.Minute) From 6915489003883335fe5be9f9a44ffc4800665419 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:02:50 -0400 Subject: [PATCH 26/70] test(integration): run internal/api's live filter test in the target TestIntegration_FilterValuesAgainstClickHouse needs a ClickHouse whose server zone is not UTC and reaches the read path's internals, so it stays beside its package, selected by name in the invocation that already runs internal/mq's integration-tagged tests (about 15s, beside them rather than after). The binding test's header now describes an in list as what it is: an external table sent as RowBinary. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- Makefile | 7 ++++--- tests/integration/query_binding_test.go | 13 +++++++------ 2 files changed, 11 insertions(+), 9 deletions(-) diff --git a/Makefile b/Makefile index 4f9f4045..863f4b9b 100644 --- a/Makefile +++ b/Makefile @@ -781,11 +781,12 @@ test-integration: go-mod-download ## Run Go integration tests + render coverage -tags="integration $(TAGS)" -timeout 480s -coverpkg=./... -race -count=1 \ ./tests/integration/... ./internal/mq/natsspike/... ./internal/cache/... $(ARGS) \ -args -test.gocoverdir="$(CURDIR)/$(COV_INT)/data" - @# internal/mq's integration-tagged tests (the external NATS broker) run - @# alone: its untagged tests are the unit suite's. + @# The integration-tagged tests of internal/mq (the external NATS broker) + @# and internal/api (the read path's filters against a ClickHouse in a + @# non-UTC zone) run alone: their untagged tests are the unit suite's. @GOCOVERDIR="$(CURDIR)/$(COV_INT)/data" go tool gotestsum --format $(GOTESTSUM_FMT) -- \ -tags="integration $(TAGS)" -timeout 240s -coverpkg=./... -race -count=1 \ - -run '^Test(ExternalNATS|NewNATS|NATSPermissions_Refuse|Leases)' ./internal/mq $(ARGS) \ + -run '^Test(ExternalNATS|NewNATS|NATSPermissions_Refuse|Leases|Integration_)' ./internal/mq ./internal/api $(ARGS) \ -args -test.gocoverdir="$(CURDIR)/$(COV_INT)/data" @if [ -z "$(COV_DEFER)" ]; then go run ./scripts/cov render integration; fi diff --git a/tests/integration/query_binding_test.go b/tests/integration/query_binding_test.go index 56e53365..05e96fee 100644 --- a/tests/integration/query_binding_test.go +++ b/tests/integration/query_binding_test.go @@ -15,16 +15,17 @@ import ( // TestStructuredQuery_FilterValuesRoundTrip drives every filter value shape // through the production /v1/query, for both the `eq` (scalar `{pN:String}`) -// and `in` (`{pN:Array(String)}`) bindings. +// and `in` (an external table of one String column, sent as RowBinary) +// bindings. // -// It exists because the two bindings need DIFFERENT encodings and a Go-side +// It exists because the two bindings carry a value DIFFERENTLY and a Go-side // unit test can only assert what the author believed: a scalar parameter is // read by ClickHouse's escaped-text reader (an unencoded backslash silently // becomes an escape sequence, an unencoded tab or newline is a hard code-457 -// parse error), while an Array(String) element is read as a quoted literal (a -// raw tab rides through, but a quote or backslash must be escaped). Applying -// either encoding to the other's value is silent data loss, so the server is -// the oracle: seed the value, filter for it, and require exactly the one row. +// parse error), while a RowBinary row carries each element's bytes as they +// are, after their length. Applying the scalar's encoding to an element, or +// leaving it off a scalar, is silent data loss, so the server is the oracle: +// seed the value, filter for it, and require exactly the one row. func TestStructuredQuery_FilterValuesRoundTrip(t *testing.T) { e := env(t) table := createTable(t, "id String, v String", "ORDER BY id") From 67e2b5d23c413292c1134201fcaa9413218733eb Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:26:36 -0400 Subject: [PATCH 27/70] build(chtypes): move to SDK go/v0.5.2 go/v0.5.2 is a drop-in from v0.5.1: same ABI revision 6, same locked artifact, so chtypes.lock is unchanged and the frozen fetch still passes. The type layer now recognises the SDK's refusal to open an image already initialised in another zone with errors.Is(err, chtypes.ErrInitConflict) instead of guessing from its zone record, so any other failure on a line open elsewhere in another zone (a library path that cannot be stat'ed, which the SDK now refuses before dlopen) keeps its own cause instead of being reported as a zone conflict. Both are still that tenant's Unavailable. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- .github/actions/setup-env/action.yml | 2 +- .github/workflows/README.md | 2 +- CHANGELOG.md | 4 +- README.md | 2 +- docs/src/content/docs/deployment.md | 10 ++--- docs/src/content/docs/development.md | 2 +- go.mod | 2 +- go.sum | 4 +- internal/typelayer/zone.go | 65 +++++++++++++++++++--------- internal/typelayer/zone_test.go | 55 +++++++++++++++++++++-- scripts/fetch-chtypes.sh | 2 +- 11 files changed, 111 insertions(+), 39 deletions(-) diff --git a/.github/actions/setup-env/action.yml b/.github/actions/setup-env/action.yml index 462721c6..f0fb45fa 100644 --- a/.github/actions/setup-env/action.yml +++ b/.github/actions/setup-env/action.yml @@ -198,7 +198,7 @@ runs: # # The SDK's default fetch dir is revision-scoped: # ~/.cache/chtypes/artifacts/abi/-. The path and the key - # prefix both name the ABI revision (6 at SDK v0.5.1), so a cache saved + # prefix both name the ABI revision (6 at SDK v0.5.2), so a cache saved # by an older SDK is never restored into the search path. When the SDK's # ABI revision changes, bump `abi6` in both places and re-lock (see # docs/development.md). diff --git a/.github/workflows/README.md b/.github/workflows/README.md index 7b88047d..83d44f73 100644 --- a/.github/workflows/README.md +++ b/.github/workflows/README.md @@ -126,7 +126,7 @@ Never add a per-job copy of content that is a pure function of a lockfile — ke `unit`, `integration` and `e2e` link the chtypes SDK (cgo dlopen of a per-ClickHouse-version `.so`/`.dylib`) and need the artifact for the line the test suite dials — today ClickHouse 26.8, matching `tests/integration/setup_test.go`'s pinned container. `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (job-level `env:` on all three) makes `typelayer.TestEngine` `t.Fatal()` if the artifact is missing instead of `t.Skip()`ing — CI must never quietly skip chtypes-backed tests. -`chtypes.lock` (repo root) pins the exact file + sha256 per platform/line; `scripts/fetch-chtypes.sh` wraps the SDK's own CLI with `--frozen --lock chtypes.lock`, so a fetch here can only install what the lock names, never the rolling `artifacts` release. `setup-env`'s `chtypes: "true"` input (see [Cache inventory](#cache-inventory)) restores `~/.cache/chtypes/artifacts/abi6` and always re-runs the fetch script afterward — cheap on a hit (a manifest check, not a re-download) and what turns a restored-but-unverified cache entry back into a hash-checked one every run. A lock is specific to the SDK's ABI revision: a build from another revision is never selected, so after an SDK bump that changes the revision (6 at v0.5.1) `--frozen` fails with `CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED` until the lock is regenerated the same way, and the `abi6` path and key prefix in `setup-env` move with it. Widening the pinned line set is a two-step: `scripts/fetch-chtypes.sh ` locally to update `chtypes.lock`, then add the line to `LOCK_LINES` in that script. Two hash mismatches are possible and they behave differently — don't read one as the other. **Upstream republished a pinned line under a new sha256**: `--frozen` refuses the artifact the lock does not name, and both `Dockerfile.goreleaser`'s fetch and CI's cache-miss fetch fail, until `chtypes.lock` is regenerated per platform (`go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch 26.8 --lock chtypes.lock --platform `, once each for `darwin-arm64`, `linux-amd64`, `linux-arm64`, without `--frozen`) — and `goreleaser-validate.yml`'s image job, which exercises this same fetch on every PR touching `chtypes.lock` or the release workflows, is what surfaces a republish at PR time rather than mid-release. **The cache holds a library the current lock no longer names** (a re-lock landed, so the exact key missed and `restore-keys` handed back the previous generation): this does *not* fail — measured, the CLI reports `is present but hashes … (want …) — replacing` and re-downloads, then the post-job save mints the new generation. So a re-lock costs one cold fetch per Go job on the first run and nothing after. +`chtypes.lock` (repo root) pins the exact file + sha256 per platform/line; `scripts/fetch-chtypes.sh` wraps the SDK's own CLI with `--frozen --lock chtypes.lock`, so a fetch here can only install what the lock names, never the rolling `artifacts` release. `setup-env`'s `chtypes: "true"` input (see [Cache inventory](#cache-inventory)) restores `~/.cache/chtypes/artifacts/abi6` and always re-runs the fetch script afterward — cheap on a hit (a manifest check, not a re-download) and what turns a restored-but-unverified cache entry back into a hash-checked one every run. A lock is specific to the SDK's ABI revision: a build from another revision is never selected, so after an SDK bump that changes the revision (6 at v0.5.2) `--frozen` fails with `CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED` until the lock is regenerated the same way, and the `abi6` path and key prefix in `setup-env` move with it. Widening the pinned line set is a two-step: `scripts/fetch-chtypes.sh ` locally to update `chtypes.lock`, then add the line to `LOCK_LINES` in that script. Two hash mismatches are possible and they behave differently — don't read one as the other. **Upstream republished a pinned line under a new sha256**: `--frozen` refuses the artifact the lock does not name, and both `Dockerfile.goreleaser`'s fetch and CI's cache-miss fetch fail, until `chtypes.lock` is regenerated per platform (`go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch 26.8 --lock chtypes.lock --platform `, once each for `darwin-arm64`, `linux-amd64`, `linux-arm64`, without `--frozen`) — and `goreleaser-validate.yml`'s image job, which exercises this same fetch on every PR touching `chtypes.lock` or the release workflows, is what surfaces a republish at PR time rather than mid-release. **The cache holds a library the current lock no longer names** (a re-lock landed, so the exact key missed and `restore-keys` handed back the previous generation): this does *not* fail — measured, the CLI reports `is present but hashes … (want …) — replacing` and re-downloads, then the post-job save mints the new generation. So a re-lock costs one cold fetch per Go job on the first run and nothing after. ## Timing (steady state, full pipeline) diff --git a/CHANGELOG.md b/CHANGELOG.md index 663d57d5..7719347d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -59,11 +59,11 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Logging goes through the `slog` default logger; no constructor takes a `*slog.Logger` anymore** (`internal/mq/embedded.go`, `internal/api/{ingest,pipes,structured_query,dlq,settings,errors,router}.go`, `internal/auth/auth.go`, `internal/discovery/{discovery,timestamp}.go`, `internal/ingest/{sweeper,worker}.go`, `internal/settings/{store,watch}.go`, `internal/chconn/chconn.go`, `internal/config/persistence.go`, `internal/app/wire.go`, `internal/testutil/logtest/` (new, + tests), `internal/testutil/testutil.go`): the general-notes refactor of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) and the cleanup deferred from [#586](https://github.com/Wave-RF/WaveHouse/pull/586), which left `internal/mq` logging half through an injected logger and half through the default. The logger parameter or field is gone from `mq.NewEmbedded`, `api.NewIngestHandler` / `NewPipesHandler` / `NewStructuredQueryHandler` / `NewDLQHandler` / `NewSettingsHandler`, `api.RequireAdmin` and `api.Dependencies.Logger`, `auth.NewAuthenticator`, `discovery.NewSchemaRegistry`, `ingest.NewSweeper` and the ingest worker, `settings.Open`, `chconn.Open` (whose field was never read), and `config.WarnIfFreshDataDir` / `LogStorageInitError`. Call sites use the context-aware calls (`slog.ErrorContext(ctx, …)`) wherever a context is in scope, so the trace handler can stamp them. Two visible differences: the ingest worker's lines no longer carry `component=ingest_worker`, and the auth middleware's operator-key audit lines and the settings reload lines are no longer skippable by passing a nil logger (only tests did). Tests reach log output through the new `internal/testutil/logtest`: `Silence()` from a package's `TestMain`, and `Capture(t, level)` for a test that asserts on log lines — which therefore runs serially, since the default logger is process-wide. `testutil.NopLogger` is removed. - **The process wiring moves out of `main.go` into `internal/app`** (`internal/app/` (new: `app.go`, `wire.go`, + tests), `cmd/wavehouse/main.go` (+ tests), `tests/integration/setup_test.go`, `.testcoverage.yml`, `.github/labeler.yml`): `app.New` builds every component from the boot config and the settings directory, `Run` drives the long-lived ones — ingest worker, sweeper, hub bridge, keepalive wheel, schema refresh, SIGHUP and the directory watcher, the API server and the Prometheus sidecar — under one `errgroup` until the signal context is cancelled or one of them fails, and `Close` releases what `New` opened in reverse order. Each component is wired in one place — what it opens, what it loops, what it releases — with the settings store handed to its wiring function whole, so the per-tenant registry ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)) lands there rather than in `main`. `main.go` shrinks to argv dispatch, the logger, `config.Load`, `CheckDataDir`, `app.New`, `app.Run`; `run(ctx)` takes the context `main` cancels on the first `SIGINT`/`SIGTERM`, so it is unit-tested end to end and the per-suite coverage exclude for it is gone. The integration suite boots through the same `app.New` against its testcontainer (the seed settings patched to the container, `default_role` set to the admin role) instead of a hand-built handler subset that had drifted from the binary. Behavior is unchanged except for the stop, which is now bounded end to end in three phases whose budgets add rather than multiply — `server.shutdown_timeout` for the drain, then fixed 5s and 3s for the release and the telemetry flush, so the worst case is the timeout plus 8s: `Run` drains the ingest worker and the API server's in-flight requests — the process could previously exit while the worker drain was still in flight — while open SSE streams are ended the moment the drain begins (a stream is a connection to close, not work to wait for; the client reconnects via `Last-Event-ID`) instead of holding the stop for the whole timeout; then `Close(ctx)` releases the stores under a context of its own, so a remote store's close can give up at the deadline rather than hang the exit, and finally flushes telemetry on a separate short budget so the flush that reports on the stop is never starved by a slow close. A settings reload caught mid-hook by the stop gives up with it. A second `SIGTERM`/`SIGINT` during the stop abandons it and exits non-zero, and `SIGHUP` is ignored once a stop has begun (it was briefly fatal: the reload loop's `signal.Stop` restored the default disposition at the start of the drain). The `mq.max_bytes_gb` reload hook bounds its JetStream calls to ten seconds, and when the DLQ resize fails it rolls the ingest stream back under a budget of its own instead of the one that just expired. `deployments/compose/standalone.yaml` sets `stop_grace_period` to cover all three phases, and the deployment docs gain a [Stopping](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md#stopping) section. Closes [#140](https://github.com/Wave-RF/WaveHouse/issues/140); story 0 of #583. -- **The chtypes SDK is `go/v0.5.1` (ABI revision 6), and the supported ClickHouse line is 26.8** (`go.mod`, `chtypes.lock`, `scripts/fetch-chtypes.sh`, `.github/actions/setup-env/action.yml`, `deployments/compose/*.yaml`, `tests/integration/setup_test.go`): the lock pins the revision-6 26.8.15.10 build (`b1790845279`) on darwin-arm64, linux-amd64 and linux-arm64, and must be regenerated whenever the SDK's ABI revision changes. The compose files, CI and the integration suite run `clickhouse/clickhouse-server:26.8.15.10`, off the retired 26.6 line. One 26.8 rule worth knowing: a bare JSON number in a `DateTime64` column is read as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — ClickHouse's rule, which WaveHouse stores as the server would. The default artifact cache moved to `~/.cache/chtypes/artifacts/abi6/-`, so the first fetch after upgrading downloads again (explicit `--dest` / `CHTYPES_REGISTRY` directories, as the Docker images use, are unaffected); CI's cache key and path carry the revision. Insert `check` clauses now cost one parse instead of two: they are judged inside the same `RowsExportWith` call that validates the body, through a compiled row filter, and the five ingest formats (JSON family, CSV, TSV, CSV and TSV with `header=present`) all take that path. +- **The chtypes SDK is `go/v0.5.2` (ABI revision 6), and the supported ClickHouse line is 26.8** (`go.mod`, `chtypes.lock`, `scripts/fetch-chtypes.sh`, `.github/actions/setup-env/action.yml`, `deployments/compose/*.yaml`, `tests/integration/setup_test.go`): the lock pins the revision-6 26.8.15.10 build (`b1790845279`) on darwin-arm64, linux-amd64 and linux-arm64, and must be regenerated whenever the SDK's ABI revision changes. The compose files, CI and the integration suite run `clickhouse/clickhouse-server:26.8.15.10`, off the retired 26.6 line. One 26.8 rule worth knowing: a bare JSON number in a `DateTime64` column is read as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — ClickHouse's rule, which WaveHouse stores as the server would. The default artifact cache moved to `~/.cache/chtypes/artifacts/abi6/-`, so the first fetch after upgrading downloads again (explicit `--dest` / `CHTYPES_REGISTRY` directories, as the Docker images use, are unaffected); CI's cache key and path carry the revision. Insert `check` clauses now cost one parse instead of two: they are judged inside the same `RowsExportWith` call that validates the body, through a compiled row filter, and the five ingest formats (JSON family, CSV, TSV, CSV and TSV with `header=present`) all take that path. - **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `builds:` and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes --notes-start-tag "$(git describe --match 'v*')"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. -- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.1, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` results — carry ClickHouse's own rendering** (`"2026-06-21 04:00:00.123"`, in the column's zone) instead of the RFC 3339 `Z`-suffixed form WaveHouse used to canonicalize to, by construction rather than by a rewriting step (closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` results — carry ClickHouse's own rendering** (`"2026-06-21 04:00:00.123"`, in the column's zone) instead of the RFC 3339 `Z`-suffixed form WaveHouse used to canonicalize to, by construction rather than by a rewriting step (closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. - **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, without the columns it may not write, instead of walking a decoded record's keys. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a checked column behaves exactly like omitting it. diff --git a/README.md b/README.md index 1914f7f2..698c7aee 100644 --- a/README.md +++ b/README.md @@ -130,7 +130,7 @@ go install github.com/Wave-RF/WaveHouse/cmd/wavehouse@latest `go install` compiles from source with cgo enabled (requires a C toolchain and glibc — Linux amd64/arm64 or macOS arm64) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse loads at start. Fetch it once before the first run: ```bash -go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch +go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch ``` This downloads 160–290 MB into the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision); point `WH_CHTYPES_REGISTRY` elsewhere if you keep it somewhere else. diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index d7a191b9..2104e0d5 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -106,7 +106,7 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ ## chtypes artifacts -**What it is.** WaveHouse validates ingest data, coerces types, substitutes `DEFAULT`s, and evaluates row-level security by running ClickHouse's own parser in-process, through [chtypes](https://github.com/wave-rf/chtypes) (`github.com/wave-rf/chtypes/go` v0.5.1). The parser itself ships as a shared library (`libchtypes.so` / `.dylib`) built per **ClickHouse minor line** (e.g. `26.8`) and per platform. There is no nearest-version fallback: each tenant's table set is compiled from its own schema refresh against the artifact matching that tenant's ClickHouse server, and a library is opened lazily, the first time a tenant on that line is bound. +**What it is.** WaveHouse validates ingest data, coerces types, substitutes `DEFAULT`s, and evaluates row-level security by running ClickHouse's own parser in-process, through [chtypes](https://github.com/wave-rf/chtypes) (`github.com/wave-rf/chtypes/go` v0.5.2). The parser itself ships as a shared library (`libchtypes.so` / `.dylib`) built per **ClickHouse minor line** (e.g. `26.8`) and per platform. There is no nearest-version fallback: each tenant's table set is compiled from its own schema refresh against the artifact matching that tenant's ClickHouse server, and a library is opened lazily, the first time a tenant on that line is bound. **The line this build is tested on.** The repository pins ClickHouse **26.8** (`chtypes.lock` names the 26.8.15.10 build, and the compose files and the integration suite run `clickhouse/clickhouse-server:26.8.15.10`). The type layer inherits ClickHouse's own parsing rules, so they are the connected server's: for instance, 26.8 reads a bare JSON number in a `DateTime64` column as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — WaveHouse stores what the server would. @@ -123,20 +123,20 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ **Release archives and `go install` / building from source** do not carry or fetch an artifact — only the Docker images bake one in. See the [README's `go install` caveat](https://github.com/Wave-RF/WaveHouse#c-go-install-binary-no-docker). Fetch one yourself before first run: ```bash -scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch --frozen --lock chtypes.lock 26.8 +scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --frozen --lock chtypes.lock 26.8 ``` or, for a line not in the repo's lock file: ```bash -go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch +go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch ``` ### Pinning with `chtypes.lock` -`chtypes.lock`, checked in at the repo root, records the exact artifact file and SHA-256 per platform/line the project builds and tests against. CI restores from it with `--frozen` (refusing anything the lock doesn't name) rather than fetching the rolling artifact release, so a pipeline never silently starts testing a new build. Refresh it deliberately — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch --lock chtypes.lock --platform `, once per platform (`darwin-arm64`, `linux-amd64`, `linux-arm64`), without `--frozen` — and commit the result; don't regenerate it implicitly. +`chtypes.lock`, checked in at the repo root, records the exact artifact file and SHA-256 per platform/line the project builds and tests against. CI restores from it with `--frozen` (refusing anything the lock doesn't name) rather than fetching the rolling artifact release, so a pipeline never silently starts testing a new build. Refresh it deliberately — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --lock chtypes.lock --platform `, once per platform (`darwin-arm64`, `linux-amd64`, `linux-arm64`), without `--frozen` — and commit the result; don't regenerate it implicitly. -A lock is specific to the SDK's ABI revision (6 at v0.5.1): the fetcher never selects a build from another revision, so after an SDK bump that changes the revision, `--frozen` fails (`CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED`) until the lock is regenerated the same way, and the CI cache key and path (`abi6`) move with it. +A lock is specific to the SDK's ABI revision (6 at v0.5.2): the fetcher never selects a build from another revision, so after an SDK bump that changes the revision, `--frozen` fails (`CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED`) until the lock is regenerated the same way, and the CI cache key and path (`abi6`) move with it. ## Releases diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index bd4bef27..7aa7af9e 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -26,7 +26,7 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: `internal/typelayer` loads a per-ClickHouse-version shared library at start to run ingest validation and row-level security through ClickHouse's own parser (see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)). It is not source code and `make tools` does not fetch it for you — pull it once with: ```bash -scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.1 fetch --frozen --lock chtypes.lock 26.8 +scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --frozen --lock chtypes.lock 26.8 ``` It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–290 MB — expect the first run to take a minute or two. Without it, `make dev` / `make test` / `make test-e2e` fail closed (a `503` on ingest, every stream row withheld) until a matching artifact exists for the ClickHouse line the tests or your local server run against. diff --git a/go.mod b/go.mod index 54741a28..c326be44 100644 --- a/go.mod +++ b/go.mod @@ -44,7 +44,7 @@ require ( github.com/samber/slog-sampling v1.7.0 github.com/stretchr/testify v1.12.1 github.com/testcontainers/testcontainers-go v0.44.0 - github.com/wave-rf/chtypes/go v0.5.1 + github.com/wave-rf/chtypes/go v0.5.2 go.opentelemetry.io/contrib/bridges/otelslog v0.20.1 go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.71.0 go.opentelemetry.io/contrib/instrumentation/runtime v0.71.0 diff --git a/go.sum b/go.sum index f3ffb6a9..fa7422e0 100644 --- a/go.sum +++ b/go.sum @@ -399,8 +399,8 @@ github.com/tklauser/numcpus v0.12.0 h1:NR85qdvHA9pFse3x3weVZ0r0ST8R6l5RHbZrlRaqo github.com/tklauser/numcpus v0.12.0/go.mod h1:ABHeXzJnr/qqwguhClkZKT1/8VABcYrsyUiUGobwWJg= github.com/vladopajic/go-test-coverage/v2 v2.18.7 h1:Kfpv8jWoC0muCAgk4bR3SvhNfqSQvO9iKZkrhdAZ4dM= github.com/vladopajic/go-test-coverage/v2 v2.18.7/go.mod h1:sTDv3QDUo3Vjo2azG9hyHfMlXH5ugnF1kqDr/4O3w4g= -github.com/wave-rf/chtypes/go v0.5.1 h1:0ke0Z2TY5yFljxMJcT9Zxmbu7QxaBeNU/bOcjoX6fQw= -github.com/wave-rf/chtypes/go v0.5.1/go.mod h1:gQE6FgwdtsXvpWlDr79q8XSwDmhiY1mIRa3gj3bOtgo= +github.com/wave-rf/chtypes/go v0.5.2 h1:Lh5yTZWMuepUcxPKcf6PuCDl41NsPBEmd/e/x3oPhpQ= +github.com/wave-rf/chtypes/go v0.5.2/go.mod h1:gQE6FgwdtsXvpWlDr79q8XSwDmhiY1mIRa3gj3bOtgo= github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e h1:JVG44RsyaB9T2KIHavMF/ppJZNG9ZpyihvCd0w101no= github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e/go.mod h1:RbqR21r5mrJuqunuUZ/Dhy/avygyECGrLceyNeo4LiM= github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU= diff --git a/internal/typelayer/zone.go b/internal/typelayer/zone.go index 58bc807f..eac1516e 100644 --- a/internal/typelayer/zone.go +++ b/internal/typelayer/zone.go @@ -1,7 +1,10 @@ package typelayer import ( + "errors" "fmt" + "slices" + "strconv" "strings" "sync" @@ -11,13 +14,14 @@ import ( // This file is the one place the server-zone rule lives. // // chtypes initialises each library in one server timezone, once per process: -// the SDK opens an artifact path at most once and refuses to open it again in -// another zone, and a per-call session_timezone does not change how a bare -// DateTime is read. So a zone is a fact about this process and one opened -// library, not about an Engine or a tenant: the first tenant to open a line's -// library fixes its zone, and a tenant on that line whose server reports -// another zone cannot be served by this process. Every library open goes -// through openLine, so the record below is complete. +// the SDK opens an artifact image (a file, however it is spelled or linked) at +// most once and refuses to open it again in another zone, and a per-call +// session_timezone does not change how a bare DateTime is read. So a zone is a +// fact about this process and one opened library, not about an Engine or a +// tenant: the first tenant to open a line's library fixes its zone, and a +// tenant on that line whose server reports another zone cannot be served by +// this process. Every library open goes through openLine, so the record below +// is complete. // // The zone reaches the SDK through its process default (SetDefaultTimezone), // not WithTimezone: a registry's WithTimezone is fixed when it is built, and @@ -27,7 +31,7 @@ import ( // zone an open reads is the one set for it. // openedZones is the zone each opened library was initialised in, keyed by the -// library's path (the SDK opens one image per path, shared by every Registry). +// path the SDK reports for it (one image per file, shared by every Registry). var openedZones = struct { mu sync.Mutex lib map[string]openedLib @@ -40,8 +44,9 @@ type openedLib struct { // openLine resolves the library for serverVersion's line, opening it in tz // when this process has not opened it yet. A non-empty cause is why the tenant -// cannot be served: no loadable artifact for the line (the SDK's own message), -// or a library already opened in another zone. +// cannot be served: no loadable artifact for the line (the SDK's own message, +// which also covers a library path that cannot be stat'ed, refused before any +// dlopen), or a library already opened in another zone. func openLine(reg *chtypes.Registry, serverVersion, tz string) (*chtypes.Library, string) { line := versionLine(serverVersion) @@ -50,16 +55,13 @@ func openLine(reg *chtypes.Registry, serverVersion, tz string) (*chtypes.Library chtypes.SetDefaultTimezone(tz) // read only if this call is what opens the library lib, err := reg.For(chtypes.Version(line)) + if errors.Is(err, chtypes.ErrInitConflict) { + // Another Registry in this process (another Engine) opened the image + // in another zone before this one resolved the line. The record only + // names that zone; the SDK's words are kept beside ours. + return nil, zoneCause(tz, line, openedZoneOfLine(line, tz)) + " (" + err.Error() + ")" + } if err != nil { - // The SDK refuses to open a path already open in another zone, which - // another Registry in this process (another Engine) reaches before it - // has resolved the line itself. Its error is untyped, so the case is - // recognised from the record, and its words are kept beside ours. - for _, o := range openedZones.lib { - if o.line == line && o.zone != tz { - return nil, zoneCause(tz, line, o.zone) + " (" + err.Error() + ")" - } - } return nil, err.Error() } o, seen := openedZones.lib[lib.Path] @@ -73,11 +75,32 @@ func openLine(reg *chtypes.Registry, serverVersion, tz string) (*chtypes.Library return lib, "" } +// openedZoneOfLine is a zone other than tz that a library of line was opened +// in, "" when the record has none. Caller holds openedZones.mu. +func openedZoneOfLine(line, tz string) string { + var zones []string + for _, o := range openedZones.lib { + if o.line == line && o.zone != tz { + zones = append(zones, o.zone) + } + } + if len(zones) == 0 { + return "" + } + return slices.Min(zones) +} + +// zoneCause is the per-tenant zone refusal; opened is "" when the zone the +// line was opened in is not known. func zoneCause(tz, line, opened string) string { + in := "another timezone" + if opened != "" { + in = strconv.Quote(opened) + } return fmt.Sprintf( - "ClickHouse reports server timezone %q, but this process opened the chtypes library for ClickHouse %s in %q; "+ + "ClickHouse reports server timezone %q, but this process opened the chtypes library for ClickHouse %s in %s; "+ "one process serves one timezone per ClickHouse version line, so serve this tenant from another process or restart", - tz, line, opened) + tz, line, in) } // versionLine is the ClickHouse line of a server version ("26.8.15.10" is diff --git a/internal/typelayer/zone_test.go b/internal/typelayer/zone_test.go index f3efcb3c..a1a3cceb 100644 --- a/internal/typelayer/zone_test.go +++ b/internal/typelayer/zone_test.go @@ -4,6 +4,8 @@ import ( "encoding/json" "fmt" "log/slog" + "os" + "path/filepath" "strings" "sync/atomic" "testing" @@ -47,9 +49,9 @@ func withMsg(recs []map[string]any, msg string) []map[string]any { // TestBind_SDKZoneRefusalIsThatTenantsUnavailable: a second Engine (its own // registry) asks for a line this process already opened in UTC, in another // zone, before its registry has resolved the line. The SDK refuses the open -// itself, with an untyped error; the tenant gets the same zone cause the -// Engine's own check gives, with the SDK's words kept, and the first Engine's -// tenant keeps answering. +// itself (ErrInitConflict); the tenant gets the same zone cause the Engine's +// own check gives, with the SDK's words kept, and the first Engine's tenant +// keeps answering. func TestBind_SDKZoneRefusalIsThatTenantsUnavailable(t *testing.T) { first := TestEngine(t, eventsTable()) // the line is open in UTC from here on @@ -72,6 +74,53 @@ func TestBind_SDKZoneRefusalIsThatTenantsUnavailable(t *testing.T) { answers(t, first, tenant.Default) } +// TestBind_UnstatableLibraryIsThatTenantsUnavailable: an artifact whose +// library path cannot be stat'ed (here a dangling symlink) is refused by the +// SDK before any dlopen. That refusal is the tenant's Unavailable in the SDK's +// words, naming the path, and it is not reported as a zone conflict although +// the line is open in another zone in this process. Once the path names the +// installed library, the same image is refused in another zone and served in +// its own. +func TestBind_UnstatableLibraryIsThatTenantsUnavailable(t *testing.T) { + first := TestEngine(t, eventsTable()) // the line is open in UTC from here on + installed := boundLib(first, tenant.Default) + require.NotNil(t, installed) + + // The explicit directory is searched first, so its copy of the test line + // shadows the installed one. + dir := t.TempDir() + line := filepath.Join(dir, testLine) + require.NoError(t, os.MkdirAll(line, 0o750)) + require.NoError(t, os.WriteFile(filepath.Join(line, "manifest.json"), + fmt.Appendf(nil, `{"library":"libchtypes.dylib","clickhouse_version":%q,"clickhouse_minor":%q}`, + TestServerVersion+"-lts", testLine), 0o600)) + link := filepath.Join(line, "libchtypes.dylib") + require.NoError(t, os.Symlink(filepath.Join(dir, "gone"), link)) + + eng, err := NewEngine(Config{RegistryDir: dir}) + require.NoError(t, err) + t.Cleanup(eng.Close) + + eng.Bind("tokyo", TestServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) + u := unavailable(t, eng, "tokyo") + assert.Empty(t, u.Table) + assert.Contains(t, u.Cause, link) + assert.Contains(t, u.Cause, "no such file or directory") + assert.NotContains(t, u.Cause, "one timezone per ClickHouse version line") + + require.NoError(t, os.Remove(link)) + require.NoError(t, os.Symlink(installed.Path, link)) + + eng.Bind("tokyo", TestServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) + u = unavailable(t, eng, "tokyo") + assert.Contains(t, u.Cause, `"UTC"`, "a symlink to the open library is the same image") + assert.Contains(t, u.Cause, "one timezone per ClickHouse version line") + + eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + answers(t, eng, tenant.Default) + assert.Same(t, installed, boundLib(eng, tenant.Default)) +} + // TestBind_ZoneRecordIsKeyedOnTheLibrary: the record names the library the // SDK opened, by path, with the zone it was opened in. func TestBind_ZoneRecordIsKeyedOnTheLibrary(t *testing.T) { diff --git a/scripts/fetch-chtypes.sh b/scripts/fetch-chtypes.sh index aa0d5427..202e7f9e 100755 --- a/scripts/fetch-chtypes.sh +++ b/scripts/fetch-chtypes.sh @@ -43,7 +43,7 @@ if [ "$(go env CGO_ENABLED 2>/dev/null || echo 0)" != "1" ]; then fi # Bump together with go.mod's `require github.com/wave-rf/chtypes/go` line. -CHTYPES_SDK_VERSION="v0.5.1" +CHTYPES_SDK_VERSION="v0.5.2" CHTYPES_CLI="github.com/wave-rf/chtypes/go/cmd/chtypes@${CHTYPES_SDK_VERSION}" REPO_ROOT="$(cd "$(dirname "$0")/.." && pwd)" From d0ba5de29e1c261ece7d02ef1e88f56494f5a430 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:31:07 -0400 Subject: [PATCH 28/70] fix(discovery): publish refreshes in the order they started Overlapping refreshes of one tenant (an on-demand refresh during an auto-refresh tick) published in the order they finished, so a refresh that read system.columns before a table was created could land after one that saw it, dropping the table from the registry and the type layer until the next refresh. The publish lock's comment claimed hooks never saw an older snapshot; that did not hold. Each refresh now takes a generation before its first read, and publishes and runs its hooks only if it started after the refresh whose snapshot is published; one that lost the race returns nil without publishing. Hooks still run before the registry is marked loaded. The integration harness no longer serializes its create-and-refresh to work around this. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- AGENTS.md | 2 +- CHANGELOG.md | 1 + docs/src/content/docs/architecture.md | 2 +- internal/discovery/discovery.go | 36 ++++++++++++--- internal/discovery/discovery_test.go | 65 ++++++++++++++++++++++++++- tests/integration/setup_test.go | 12 +---- 6 files changed, 97 insertions(+), 21 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index badd6366..a8eea4dd 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -37,7 +37,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`config/`** — YAML + env var config loading (cleanenv); strict on both sides (undeclared YAML key, unbound `WH_*` variable) and probes `data_dir` writability when a selected backend keeps state there (`NeedsDataDir`); `backends.go` holds each layer's `.backend` (the in-process value by default; `mq.backend` also takes `nats`, with its `mq.nats` sub-block of file-path-only credentials; `coord.backend` takes `nats`, whose `coord.nats` block names only the lease bucket and rides `mq.nats`'s connection (both blocks, their rules and warnings are `mq_nats.go`); `cache.backend` takes `redis`, whose sub-block is `cache_redis.go`; and `dedupe.backend` takes `dynamodb`, with its `dedupe.dynamodb` sub-block) and `Warnings`, the valid combinations boot logs at `WARN`; `config.go` holds `roles` (`Has(Role)`) and `instance_id`, and `Validate` refuses a role split the backends cannot serve (any split over the embedded MQ; `api` without `ingest`, or the reverse, over a local cache; `coord.backend=nats` without `mq.backend=nats`; `mq.backend=nats` with `coord.backend=local` in a process running `ingest`; a process running only `sweeper` under `mq.backend=nats`) — boot is the validator, there is no dry run - **`coord/`** — leases for work that must run in one process at a time (`Observer.Held` reads whether one is held without campaigning): `Coordinator.TryAcquire(ctx, name)` → a `Term` (fencing `Token`, strictly increasing per name; `Done`/`Err`, `ErrLost` on loss; `Resign`), `ErrHeld` while another holder's — or this coordinator's own — term is live; `RunElected` runs a loop only while holding its lease, resigning when the loop returns and campaigning again every `RetryPeriod`. `Local` is the in-process implementation (first taker wins, never expires; `Peer` is a second handle over the same table for tests); every implementation runs `coordtest.Conformance`. Imports only the standard library, so a distributed backend lives beside its connection: `coord.backend: nats` is `internal/mq/lease.go` (`ExternalNATS.Leases`), a key per lease in the operator's KV bucket, the KV revision as the fencing token, and expiry judged on the candidate's own clock (the same revision seen unchanged for 15s), never by a server TTL. `internal/app`'s `wireCoord` opens the one `coord.backend` selects and the sweeper runs through `RunElected` under the `sweeper` lease - **`dedupe/`** — `Deduplicator` interface (two-phase `Reserve`/`Commit`/`Release` over `Key{Table, ID}`; every backend passes the `dedupetest` conformance suite) → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant and table, pending claims in memory, committed ids stored with their expiry and deleted by an hourly background sweep along with the version-0 keys from before the table joined the key, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges) or `Dynamo` (one shared DynamoDB table, conditional `PutItem` claims; conformance-tested against dynamodb-local, selected by `dedupe.backend: dynamodb`; boot checks the table and never creates it outside dynamodb-local), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` or, gated on the table check (`Factory.Gated`), `Dynamo.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) -- **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`, and starts a tenant over on a fresh registry when a reload moves it to another address or database: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version and default timezone, and fires an `OnRefresh` hook after every successful refresh and before the registry reports itself loaded, so a loaded tenant is a bound one — `typelayer.Engine.Bind` is its only consumer (Key Design Decision #21) +- **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`, and starts a tenant over on a fresh registry when a reload moves it to another address or database: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version and default timezone, and fires an `OnRefresh` hook after every refresh it publishes and before the registry reports itself loaded (overlapping refreshes publish in the order they started: one that finishes after a later-started one has published is dropped, so an older snapshot never replaces a newer one), so a loaded tenant is a bound one — `typelayer.Engine.Bind` is its only consumer (Key Design Decision #21) - **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else; `row` is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced for that stored record (via `typelayer`'s `Table.IngestWith`), `columns` names its positions (the table's insertable columns, or the narrower list a column-restricted role produces); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs with `typelayer.InsertSettings()` plus `async_insert=0` (it never loads the artifact, so an ingest-only process needs none). In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) - **`keyenc/`** — the one escaping composite keys are built from: `Escape` keeps `[A-Za-z0-9_-]` (exactly the tenant-id grammar, so a tenant id is its own escaped form) and writes every other byte as `%XX`, `Unescape` is `url.PathUnescape` (lenient: either hex case, and a byte left unescaped reads as itself, so a `%2D` an earlier build wrote still reads), `Join`/`AppendJoin` escape each field and put a separator between them (they panic on no fields, and on a separator the escaping could write or one outside ASCII) and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to and no field reaches a key unescaped: NATS subjects (`Join`/`Split` after the verbatim tenant) and the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — and the dedupe keys (`/
/`) use it; changing what it keeps orphans every stored key (an orphaned dedupe key lets a seen id through again), and on the shared backend, whose keys every process builds for itself, splits them between builds for the length of a rolling upgrade (a bump one build makes misses the entries the other filed, served until their TTL) - **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20), and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal, `ErrUnavailable` a broker that cannot be reached — both a `503`, with `Retry-After` `30` and `5`; `WithIdempotencyKey` makes a republish inside the queue's duplicate window a no-op), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). A broker whose ingest queue is split into units one consumer at a time owns implements `Sharded` too (`IngestUnits`, `ResetOrphaned`, `Unowned`; `ConsumerConfig.Units` narrows a consumer to some of them, and its consumer is a `Releaser` and a `Halter`, capped per unit by `ConsumerConfig.MaxHeld`, with `ErrConsumerMismatch` when an operator durable no longer fits), and `Message.OnSettled` runs a hook once, at the first ack or nak attempt, confirmed or not. Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, stream names, sequences, and ack floors are private to the implementations, whose subject tokens are escaped by the shared `internal/keyenc`: `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`, `deadletter.go`), which `internal/app` constructs and hands everything else as a `mq.Broker`, and `ExternalNATS` (`external.go`, `subject_nats.go`, `nats_topology.go`: an operator-owned cluster whose streams, durables and lease bucket it never creates, changes, purges or deletes; `lease.go` holds `coord.backend: nats`'s leases in that bucket), which `internal/app` constructs from the `mq.nats` block when `mq.backend` is `nats` ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every implementation passes the conformance suite in `internal/mq/mqtest` (`mqtest.Run`), which states the `Broker` contract as behavior; a new backend runs it from its own test, with `mqtest.Caps` only where its semantics legitimately differ diff --git a/CHANGELOG.md b/CHANGELOG.md index 7719347d..fab146cc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -113,6 +113,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **A timestamp filter on a column with a time zone matches the instant it names** (`internal/query/{builder,bind}.go` (`bind.go` new; + tests), `internal/api/{clickhouse_http,structured_query}.go` (+ tests, + integration-tagged `structured_query_integration_test.go`), `tests/e2e/sdk/query.test.ts`, `docs/src/content/docs/{api,architecture}.md`): `POST /v1/query` rewrote an RFC 3339 filter value, and formatted `time_range` bounds, as zone-less UTC text that ClickHouse then read in the **column's** zone — so on a `DateTime('Asia/Tokyo')` column `eq "2026-06-21T04:00:00Z"` matched nothing, on `DateTime64(3, 'America/New_York')` it matched the row four hours later, and on a zone-less column under a non-UTC server the row the server's offset away (measured on ClickHouse 26.8.15.10 and 24.8.14.39). WaveHouse no longer rewrites the value: on a `Date`/`DateTime`/`DateTime64` column it binds as written and ClickHouse parses it with `parseDateTime64BestEffort` in the column's declared zone, so an offset or `Z` is the exact instant, a zone-less value still means the column's local time, a fraction compares exactly against a whole-second column (ClickHouse refused it as a type mismatch before), and a value ClickHouse cannot parse is `400 clickhouse.rejected`. The primary key is still used. `time_range` bounds go as RFC 3339 in UTC through the same parse. **An `in` list is no longer capped**: it went as one query-string parameter, which ClickHouse caps at 128 KiB, and answered `400` past that; it now travels as an external table in a multipart body, so any list the 1 MiB request carries reaches ClickHouse, compared under the column's type and with the primary key in use. One difference from the old list: an element that is not a value of the column's type (`"1.5"` on an integer column) matches no row instead of failing the query. The limits left are ClickHouse's own, each a `400` that names it: one scalar value over 128 KiB once URL-encoded, all of them over the request line, and — for a query with an `in` list — SQL over the 128 KiB form field it travels in. +- **A schema refresh that started first can no longer publish last** (`internal/discovery/discovery.go` (+ tests), `tests/integration/setup_test.go`, `AGENTS.md`, `docs/src/content/docs/architecture.md`): overlapping refreshes of one tenant, such as `POST /v1/ops/schema/refresh` during an auto-refresh tick, published in the order they finished, so a refresh that read `system.columns` before a table was created could land after one that saw it and drop the table from the registry and the type layer until the next refresh. Each refresh now takes a generation before its first read and publishes, and runs its hooks, only if it started after the refresh whose snapshot is published; one that lost the race returns success without publishing. - **A tenant moved to another ClickHouse address or database discovers its new schema with the reload** (`internal/app/{wire,discoveries}.go` (+ tests), `internal/discovery/discovery.go` (comment), `internal/testutil/testutil.go`, `docs/src/content/docs/architecture.md`, `settings-directory.mdx`, `AGENTS.md`): closes [#638](https://github.com/Wave-RF/WaveHouse/issues/638). A reload that changed a tenant's `clickhouse.addr` or `clickhouse.database` orphaned the tenant's cache and left its schema registry as it was, so until the tenant's loop fired at `schema.refresh_interval` its queries and inserts were validated against the previous database's schema and run against the new one. The tenants the pools reconcile reports stale now have their registry dropped in the same hook, and the discovery reconcile that follows builds each a fresh one over the pool it is on, as it does for a tenant back after a rejection or removal: the first discovery runs at once in the tenant's own loop, so the reload never waits on ClickHouse, and until it succeeds the tenant's table lookups answer `503` with `Retry-After: 5`, as before any first discovery. A discovery that fails is logged with its tenant (`schema discovery retry failed`), counted in `wavehouse_schema_refresh_failures_total`, and retried with backoff from two seconds to sixty. A tenant whose address and database did not change keeps its registry and its loop, a flat directory's tenant `0` moves the same way, and a process without the api role, which discovers no schema, only repoints its pool. - **A shard's owner keeps it while its ClickHouse is slow, one stuck shard no longer stalls the process's others, and a handover or clean stop keeps each table's rows in order** (`internal/mq/external_consumer.go` (new, from `external.go`), `internal/mq/{external,mq}.go`, `internal/ingest/{claims,worker}.go`, `.testcoverage.yml`, tests in `internal/mq/external_test.go`, `internal/ingest/{claims,worker}_test.go` and `tests/integration/shard_order_test.go` (new), `docs/src/content/docs/{deployment,ingest-pipeline,architecture}.md`, `AGENTS.md`): part of the external-NATS workstream of [#613](https://github.com/Wave-RF/WaveHouse/issues/613), fixing the shard ownership of the entries above before they ship. The pull consumer ran the worker's handler on its delivery goroutine and pulled again only when it returned, and the server renews a shard's pin only on a pull, so a process at its cap of held rows for longer than the pinned TTL (10 seconds) lost its pins: `wavehouse_ingest_shards_unowned` counted its shards, and a peer that judged it dead reset them, redelivering rows it still held — two writers per table. Each shard now has a puller that never runs the handler, fetches only what the shard's cap leaves room for, and at the cap renews the pin every 5 seconds with a one-row fetch, at most one `ack_wait` of such rows past the cap, 12 at the generated 1 minute (past that, and once halted, with a pull of `max_bytes` 1, which delivers nothing); measured, an owner blocked 13 seconds on a hung insert kept one pin throughout and took 2 rows past its share, and a unit drains as fast as before (160,000–192,000 rows a second, against 158,000–172,000). `ResetOrphaned` and `Unowned` also require that the shard delivered nothing and had no ack for its pinned TTL. The process-wide cap of 10,000 held rows let one stuck table take every slot and stall every shard the process owned; each shard now holds its own share, plus at most the 12 rows its renewals take (see the entry above), and with one shard's share full a healthy shard's rows kept flowing. A clean stop now halts every shard first (`mq.Halter`), keeping the pins, so the worker writes what it holds and what the shards still deliver, and releases them after its final flush; rows delivered during that flush were dropped before, and came back only once they had been waited out. A handover keeps the pin until the rows delivered are written, bounded by the worker's 60-second ack wait instead of 15 seconds, and a row the worker drops while stopping is NAKed before the release. Checked under continuous publishing, with the first process's ClickHouse slower than a batch window: every table's rows arrived exactly once and in order across a handover and a clean stop, and the next owner wrote its first rows about 2 seconds after the stop, against 9 seconds when the stop leaves rows behind. A shard fetches in 1-second pulls, so a shard stops fetching within a second of a halt, and an idle shard costs one pull a second. A shard durable deleted while no pull of it was waiting stalled silently until the five-minute topology check, since the next pull only got no responders: two such pulls in a row, or one renewal, now look the durable up, and one that is gone, or whose stream is, ends the worker (measured: 5 seconds after the delete, with the handler busy). A bind whose durable or stream is gone, or whose durable no longer fits (`mq.ErrConsumerMismatch`: `ack_wait` shorter than asked, or no `max_ack_pending`), ends the worker instead of retrying every tick with a warning, and a bind failure that lasts a minute logs an error. `wavehouse_ingest_rows_held_waits_total` is gone: nothing waits at the cap any more. A renewal at the cap takes a row rather than holding it back: a held-back redelivery goes to the back of the server's redelivery queue, and with max_bytes-1 renewals rows NAKed as 0, 1, 2, 3 came back as 2, 3, 0, 1 (measured); now in order, until a stuck shard is 12 rows past its cap. The server requeues the same way while another process's pull waits for the pin, so rows redelivered across a handover or a stop may come back out of order among themselves, though still ahead of newer rows. The unpin on release names no pin id, so it is preceded by a check that the pin is still this process's; the unit renews its pin until the unpin returns (for up to `ack_wait` and one 5-second renewal after its halt has handed on what it fetched, longer than a handover waits), so the pin cannot lapse between the two, and only a consumer leader change there could move it. - **External NATS: the shipped manifests fit the shipped volume, boot refuses permissions narrower than the shards, and the tooling covers `js_domain`** (`internal/mq/{nats_manifests,nats_topology,external}.go`, `internal/mq/nats_permprobe.go` (new), `internal/mq/natstest/natstest.go`, `cmd/wavehouse/mq.go` (+ tests; `internal/mq/{nats_manifests,external_perms}_test.go` new), `deployments/nats/{jetstream.yaml,values.yaml}`, `docs/src/content/docs/{deployment,architecture}.md`, `configuration.mdx`): part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). The shipped streams reserved 225 GiB against the shipped 100 Gi volume, whose size the NATS Helm chart makes each server's `max_file_store`, so the server refused the third partition. `wavehouse mq manifests --file-store` (default `100Gi`) now sizes each partition at 15% of it, the history at 10% and the dead-letter stream at 5%, so the shipped four partitions reserve 75 GiB; a partition's size does not follow N, and the generator refuses streams that reserve more than the store (`--partition-max-bytes` resizes them). The test server takes its file store from the shipped values as the chart does, where it had a limit no test could reach. Boot now probes every shard durable as the connecting user, with a pull and an unpin the server rejects on their merits, and a `required` finding names each durable the user may not pull, where permissions generated for fewer shards passed boot and left those shards' rows unpulled; a consumer request the server refuses later sets `wavehouse_mq_topology_ok` to `0` at once. `wavehouse mq permissions --js-domain` allows and denies the `$JS..API` subjects a client in a JetStream domain sends, beside the plain ones a server in the domain checks after mapping them; `wavehouse mq manifests` takes `--ingest-consumer`, `--history-stream` and `--publish-timeout`, the duplicate window following the timeout; the `wavehouse` user may ack only for its shard durables, under either ack subject layout, where it could ack for any consumer; and an `mq.nats.ingest_consumer` or `history_stream` outside `[a-zA-Z0-9_-]` refuses boot. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index c09bcff6..a99acda0 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -158,7 +158,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `discovery/` — Schema Discovery -- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant to another user or `tls` block, which reads the same tables, applies to the next refresh, one that moves it to another address or database has `internal/app`'s `discoveries` drop its registry and build a fresh one, whose first discovery runs at once and whose lookups answer `ErrNotLoaded` until it succeeds ([#638](https://github.com/Wave-RF/WaveHouse/issues/638)), and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) can decide the insertable subset the ingest envelope and the SSE announcement are both built from. Each refresh also records the server version (`SELECT version()`) and default time zone (`SELECT timezone()`, exposed via `ServerTimezone()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), and reads each column's `default_expression` and 1-based `position` alongside its type. An `OnRefresh` hook fires synchronously after every successful refresh and **before** the registry reports itself loaded, so a loaded tenant is a bound one — `internal/typelayer.Engine.Bind` is its only registered consumer, and it is what resolves the chtypes artifact matching that tenant's server line and recompiles its per-table handles ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop, each sleep drawn uniformly below the backoff so instances retrying one ClickHouse do not retry in lockstep, used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. +- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant to another user or `tls` block, which reads the same tables, applies to the next refresh, one that moves it to another address or database has `internal/app`'s `discoveries` drop its registry and build a fresh one, whose first discovery runs at once and whose lookups answer `ErrNotLoaded` until it succeeds ([#638](https://github.com/Wave-RF/WaveHouse/issues/638)), and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) can decide the insertable subset the ingest envelope and the SSE announcement are both built from. Each refresh also records the server version (`SELECT version()`) and default time zone (`SELECT timezone()`, exposed via `ServerTimezone()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), and reads each column's `default_expression` and 1-based `position` alongside its type. Overlapping refreshes (the loop and an on-demand one) publish in the order they started: a refresh that finishes after a later-started one has published returns without publishing, so an older snapshot never replaces a newer one. An `OnRefresh` hook fires synchronously after every refresh that publishes and **before** the registry reports itself loaded, so a loaded tenant is a bound one — `internal/typelayer.Engine.Bind` is its only registered consumer, and it is what resolves the chtypes artifact matching that tenant's server line and recompiles its per-table handles ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop, each sleep drawn uniformly below the backoff so instances retrying one ClickHouse do not retry in lockstep, used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. - **discovery_test.go** — Unit tests for schema discovery. ### `typelayer/` — In-Process ClickHouse Parser diff --git a/internal/discovery/discovery.go b/internal/discovery/discovery.go index 529cb92c..2a6eab22 100644 --- a/internal/discovery/discovery.go +++ b/internal/discovery/discovery.go @@ -224,11 +224,19 @@ type SchemaRegistry struct { // onRefresh are the hooks a successful Refresh runs with what it // published; registered before the first Refresh, guarded by mu. onRefresh []RefreshHook - // publishMu orders a Refresh's publish, its hooks and the loaded flag, so - // two overlapping refreshes (the loop and a manual one) run their hooks in - // the order they published and a hook never sees a snapshot older than - // the one it is replacing. - publishMu sync.Mutex + // readGen numbers refreshes in the order they start, before any read. + readGen atomic.Uint64 + // publishMu makes a Refresh's publish, its hooks and the loaded flag one + // step, and guards publishedGen, the readGen of the snapshot published + // last. A refresh publishes only if it started after that one, so when + // two overlap (the loop and a manual one) and the one that started first + // finishes last, its snapshot — possibly read before a table the other + // saw was created — is dropped rather than replacing the other's. Hooks + // therefore run one refresh at a time, in start order, and are never + // handed a snapshot from a refresh that started before the one whose + // snapshot they are replacing. + publishMu sync.Mutex + publishedGen uint64 } // RefreshHook is told what a successful Refresh published: the server's @@ -264,7 +272,8 @@ func NewSchemaRegistry(source Source, id tenant.ID, refreshInterval func(tenant. // layer binds here). Hooks run synchronously on the refreshing goroutine, in // registration order, and must be registered before the first Refresh. A // failed Refresh runs none: the previous schemas stay, and so does whatever -// the hooks built from them. +// the hooks built from them. Neither does a Refresh that a later-started one +// has already published past. func (sr *SchemaRegistry) OnRefresh(hook RefreshHook) { sr.mu.Lock() defer sr.mu.Unlock() @@ -274,12 +283,20 @@ func (sr *SchemaRegistry) OnRefresh(hook RefreshHook) { // Refresh rebuilds the in-memory schema cache: it discovers the server's default // time zone and version, queries system.columns, attaches each table's DDL from // system.tables, precomputes timestamp column specs, and then runs the -// OnRefresh hooks before marking the registry loaded. +// OnRefresh hooks before marking the registry loaded. A refresh that started +// before the one whose snapshot is already published returns nil without +// publishing: the published refresh started later, so it saw everything +// committed before this one started. func (sr *SchemaRegistry) Refresh(ctx context.Context) error { tracer := otel.GetTracerProvider().Tracer("wavehouse-discovery") ctx, span := tracer.Start(ctx, "SchemaRegistry.Refresh") defer span.End() + // Taken before the first read, so a refresh that starts after a table is + // created has a higher generation than every refresh that could have + // read the database without it. + gen := sr.readGen.Add(1) + // One connection and one database per refresh, read together: a reload // that moves the tenant to another pool or database applies to the NEXT // refresh, so every query of this one runs against the same server and @@ -370,6 +387,11 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { sr.publishMu.Lock() defer sr.publishMu.Unlock() + if gen < sr.publishedGen { + slog.DebugContext(ctx, "schema refresh superseded by a later one; not published", "tenant", sr.tenant) + return nil + } + sr.publishedGen = gen sr.mu.Lock() sr.tables = tables sr.serverVersion = serverVersion diff --git a/internal/discovery/discovery_test.go b/internal/discovery/discovery_test.go index 862f349e..601f75ce 100644 --- a/internal/discovery/discovery_test.go +++ b/internal/discovery/discovery_test.go @@ -5,6 +5,7 @@ import ( "encoding/json" "errors" "log/slog" + "slices" "strings" "sync/atomic" "testing" @@ -541,7 +542,8 @@ func TestOnRefresh_HooksRunInRegistrationOrder(t *testing.T) { // TestOnRefresh_OverlappingRefreshesDoNotInterleaveHooks: the manual refresh // can overlap the auto-refresh loop. Each refresh's publish and hooks run as -// one step, so a hook never runs concurrently with another refresh's hook. +// one step, so a hook never runs concurrently with another refresh's hook. A +// refresh that finishes after a later-started one published runs none. func TestOnRefresh_OverlappingRefreshesDoNotInterleaveHooks(t *testing.T) { t.Parallel() conn := &fakeConn{columns: []fakeColumn{{table: "events", name: "id", chType: "UInt64", position: 1}}} @@ -568,10 +570,69 @@ func TestOnRefresh_OverlappingRefreshesDoNotInterleaveHooks(t *testing.T) { for range refreshes { require.NoError(t, <-errs) } - assert.Equal(t, int32(refreshes), calls.Load()) + assert.GreaterOrEqual(t, calls.Load(), int32(1)) + assert.LessOrEqual(t, calls.Load(), int32(refreshes)) assert.Equal(t, int32(1), maxInHook.Load(), "hooks of overlapping refreshes must not interleave") } +// gatedConn is a fakeConn whose system.columns query announces that it has +// started, then waits for release. +type gatedConn struct { + *fakeConn + reading chan<- struct{} + release <-chan struct{} +} + +func (c gatedConn) Query(ctx context.Context, q string, args ...any) (driver.Rows, error) { + if strings.Contains(q, "system.columns") { + c.reading <- struct{}{} + <-c.release + } + return c.fakeConn.Query(ctx, q, args...) +} + +// TestRefresh_OlderSnapshotFinishingLastIsNotPublished: a refresh that read +// the database before a table was created (the loop) finishes after one that +// started later and saw the table (a manual refresh). The newer snapshot +// stays published, and the hook is never handed the older one, so the table +// does not drop out of the registry or the type layer until the next refresh. +func TestRefresh_OlderSnapshotFinishingLastIsNotPublished(t *testing.T) { + t.Parallel() + events := fakeColumn{table: "events", name: "id", chType: "UInt64", position: 1} + created := fakeColumn{table: "created", name: "id", chType: "UInt64", position: 1} + reading, release := make(chan struct{}), make(chan struct{}) + older := gatedConn{fakeConn: &fakeConn{columns: []fakeColumn{events}}, reading: reading, release: release} + newer := &fakeConn{columns: []fakeColumn{created, events}} + + sources := []driver.Conn{older, newer} + var sourced atomic.Int32 + sr := NewSchemaRegistry(func() (driver.Conn, string) { return sources[sourced.Add(1)-1], "test" }, + tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + var hooked [][]string + sr.OnRefresh(func(_, _ string, tables []*TableSchema) { + names := make([]string, 0, len(tables)) + for _, ts := range tables { + names = append(names, ts.Name) + } + slices.Sort(names) + hooked = append(hooked, names) + }) + + olderDone := make(chan error, 1) + go func() { olderDone <- sr.Refresh(context.Background()) }() + <-reading // the older refresh has started and is reading system.columns + + require.NoError(t, sr.Refresh(context.Background())) + require.NotNil(t, sr.Get("created")) + + close(release) + require.NoError(t, <-olderDone, "a superseded refresh is not a failure") + + assert.NotNil(t, sr.Get("created"), "the older snapshot must not replace the newer one") + assert.Equal(t, [][]string{{"created", "events"}}, hooked, "the hook only ever sees the newer snapshot") + assert.True(t, sr.Loaded()) +} + // TestRefresh_WarnsOnceForTablesWithoutDDL: the two scans are not one // snapshot, so a table can be discovered without its CREATE statement. That // is one warning per refresh naming every such table, not one per table. diff --git a/tests/integration/setup_test.go b/tests/integration/setup_test.go index 54593d04..88891868 100644 --- a/tests/integration/setup_test.go +++ b/tests/integration/setup_test.go @@ -96,17 +96,11 @@ func env(t *testing.T) *testEnv { // (or sequentially-run) tests don't collide on table state. var tableCounter atomic.Uint64 -// createMu serializes createTable's create-and-refresh. -var createMu sync.Mutex - // createTable creates a uniquely-named ClickHouse table for the calling test // and registers cleanup to drop it. The schema registry is refreshed after // creation so the API discovers the new table. Returns the table name. -// -// Parallel tests create tables concurrently, and two overlapping refreshes -// publish in the order they finish, not the order they read system.columns: -// an older snapshot landing last would drop the newer table from the -// registry. createMu makes each create-and-refresh one step. +// Parallel tests create tables concurrently: the registry never lets a +// refresh that started before this one replace what this one published. // // Pass the column DDL fragment without the wrapping `()` — for example: // @@ -123,8 +117,6 @@ func createTable(t *testing.T, columns, tableOpts string) string { "CREATE TABLE IF NOT EXISTS %s (%s) ENGINE = MergeTree() %s", name, columns, tableOpts, ) - createMu.Lock() - defer createMu.Unlock() if err := sharedEnv.chConn.Exec(ctx, stmt); err != nil { t.Fatalf("create test table %s: %v", name, err) } From 2dd4de7b2211e0cc848fceb02d5f66eb0f346686 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:32:22 -0400 Subject: [PATCH 29/70] test(integration): clear gosec's G703 on the settings cleanup golangci-lint with the integration build tag flagged removeTestSettings' RemoveAll as a path traversal. Its path is always the scratch directory writeTestSettings made with os.MkdirTemp; say so where the linter reads it. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- tests/integration/setup_test.go | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/integration/setup_test.go b/tests/integration/setup_test.go index 88891868..01d02d24 100644 --- a/tests/integration/setup_test.go +++ b/tests/integration/setup_test.go @@ -285,7 +285,9 @@ func writeTestSettings(ch *chInstance) (string, error) { // removeTestSettings removes a directory writeTestSettings made, with its // scratch parent. -func removeTestSettings(dir string) { _ = os.RemoveAll(filepath.Dir(dir)) } +func removeTestSettings(dir string) { + _ = os.RemoveAll(filepath.Dir(dir)) //nolint:gosec // G703: dir is /settings from writeTestSettings +} // tenantSettings is one tenant's four files: the seed with the ClickHouse // block pointed at the testcontainer's database, and the dev-style policy. From d02cb781d9e22b901558d3296e4d5f02ce290719 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:45:13 -0400 Subject: [PATCH 30/70] docs: fix review-round findings across the chtypes docs Correct the query-path error tables and 64 MiB cap, restore the EPHEMERAL, timestamp-rendering and read-connection-cap wording for the intended behaviour, document chtypes_registry and the 26.8-only images, and sync AGENTS.md, CHANGELOG, README, SECURITY and deployment comments. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- AGENTS.md | 4 +- CHANGELOG.md | 12 +++--- README.md | 6 ++- SECURITY.md | 4 +- deployments/Dockerfile | 2 +- deployments/compose/standalone.yaml | 5 ++- docs/src/content/docs/access-control.mdx | 6 +-- docs/src/content/docs/api.md | 42 ++++++++++---------- docs/src/content/docs/architecture.md | 18 +++++---- docs/src/content/docs/configuration.mdx | 11 +++-- docs/src/content/docs/deployment.md | 21 +++++----- docs/src/content/docs/development.md | 6 +-- docs/src/content/docs/getting-started.md | 2 +- docs/src/content/docs/index.mdx | 8 ++-- docs/src/content/docs/pipes.mdx | 2 +- docs/src/content/docs/reverse-proxy.mdx | 2 +- docs/src/content/docs/sdk/reference.md | 6 +-- docs/src/content/docs/sdk/streaming.md | 2 +- docs/src/content/docs/settings-directory.mdx | 10 ++--- docs/src/content/docs/why-wavehouse.md | 4 +- 20 files changed, 92 insertions(+), 81 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index a8eea4dd..a1091be6 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -47,7 +47,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes - **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row via `typelayer`, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes` (the sole exception: `cmd/wavehouse/main.go` references `typelayer` itself). One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns, plus a `DEFAULT ''` per `_eq` check column — which is how column policy and auto-inject are answered with no Go-side record inspection. `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) +- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns, plus a `DEFAULT ''` per `_eq` check column — which is how column policy and auto-inject are answered with no Go-side record inspection. `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) - **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -119,7 +119,7 @@ Verbose: `V=1 make test`. Extra args: `make test ARGS="-run TestFoo"`. Build tag Tooling notes (the non-obvious bits `make help` won't tell you): - Dev tools (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `go-test-coverage`, `gocover-cobertura`, `deadcode`, `gsa`, `goda`) are pinned in `go.mod` via `tool` directives — `go tool `, no manual install. -- `golangci-lint` is pinned in the Makefile (v2.11.4), auto-installed to `.bin/` on first `make lint` — kept out of `go.mod` (its deps conflict with the main module). +- `golangci-lint` is pinned in the Makefile (v2.13.2), auto-installed to `.bin/` on first `make lint` — kept out of `go.mod` (its deps conflict with the main module). - `pnpm` (≥ 11.21) + `Node 22 LTS` (`.nvmrc`, matches CI) must be on PATH; `make tools` runs one root `pnpm install --frozen-lockfile` across the three workspaces (SDK `clients/ts/`, E2E `tests/e2e/sdk/`, docs `docs/`). - **GNU Make 4+** required (uses `--output-sync=target`); macOS BSD Make 3.81 won't parse it. Full setup: `docs/src/content/docs/development.md` § Prerequisites. - **Lint split**: Biome owns JS/TS/JSON, markdownlint owns Markdown *and MDX* style — including two repo-local rules, WH001 (no hard-wrapped prose) and WH002 (MDX fence beside a JSX tag) in `scripts/markdownlint-rules/` — misspell owns spelling (all under `make lint`/`make fix`); accuracy/clarity/doc-sync is the `docs-reviewer` gate (§Docs review). See §Markdown authoring rules. diff --git a/CHANGELOG.md b/CHANGELOG.md index fab146cc..cf08e350 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -63,11 +63,11 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `builds:` and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes --notes-start-tag "$(git describe --match 'v*')"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. -- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` results — carry ClickHouse's own rendering** (`"2026-06-21 04:00:00.123"`, in the column's zone) instead of the RFC 3339 `Z`-suffixed form WaveHouse used to canonicalize to, by construction rather than by a rewriting step (closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. -- **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, without the columns it may not write, instead of walking a decoded record's keys. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a checked column behaves exactly like omitting it. +- **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, without the columns it may not write, instead of walking a decoded record's keys. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a checked column behaves exactly like omitting it. An `_eq` check on a column the role may not write is still injected and published. A role whose column policy cannot be compiled is refused with a non-retryable error and the cause is logged, rather than answered with a `503` that a retry could not fix. -- **WaveHouse now requires cgo, and supported platforms narrow to darwin/arm64, linux/amd64, linux/arm64** (BREAKING; `go.mod`, `scripts/build.sh`, `.goreleaser.yaml`, `deployments/Dockerfile`, `deployments/Dockerfile.goreleaser`, `Makefile`, `internal/config/config.go`, `config.yaml`, `cmd/wavehouse/main.go`, `chtypes.lock` (new), `scripts/fetch-chtypes.sh` (new), `.github/actions/setup-env/action.yml`, `.github/workflows/ci.yml`): the native type layer above needs cgo for `dlfcn` (no C library linked, no header). cgo is now unconditional: `CGO_ENABLED=0` is gone from every build path, and the `make audit-cgo` target that policed the old no-cgo build has been **removed** along with it (`make binary-analysis` is now `size` + `deadcode`). Because chtypes publishes artifacts only for darwin-arm64, linux-amd64 and linux-arm64, **Windows, FreeBSD and darwin/amd64 builds are discontinued** — `.goreleaser.yaml`'s matrix drops from 8 targets to 3, and the release archives/checksums/GHCR image narrow to match. The runtime image moves from an Alpine/musl builder + `distroless/static` to `golang:1.27-bookworm` (glibc, ships gcc) + `distroless/cc-debian12` (glibc + libstdc++, which the SDK's shared library needs) and bakes the pinned chtypes artifact into the image at `/opt/chtypes/artifacts` via a new `chtypes.lock` (exact file + sha256 per platform/line) and `scripts/fetch-chtypes.sh --frozen` wrapper, so the container has no first-request download. `go.mod` moves to `go 1.27`. Only processes running the `api` role load the artifact, so an ingest-worker-only or sweeper-only process needs none installed; the glibc requirement is 2.34 or later. New boot config: `clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY` lets an operator point at an explicit registry directory instead of the SDK's own search path (the shipped image instead sets the SDK's own `CHTYPES_REGISTRY` env var directly). CI's `unit`/`integration`/`e2e` jobs fetch and cache the pinned artifact (`setup-env`'s new `chtypes` input) and set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` so a missing artifact fails the job instead of silently skipping the chtypes-backed tests. `GOLANGCI_LINT_VERSION` bumped `v2.11.4` → `v2.13.2`: the `go 1.27` bump panics `v2.11.4`'s type checker on every package; `v2.13.0` is the oldest release whose changelog claims go1.27 support, but it panics in this tree for a different reason (`nilness`/`honnef.co/go/tools@v0.8.0-rc.1` crashing while analyzing a third-party dependency), fixed once that dependency moves past its release candidate in `v2.13.1`. The cross-toolchain approach this bullet originally described was replaced before landing — see the release-pipeline entry above. +- **WaveHouse now requires cgo, and supported platforms narrow to darwin/arm64, linux/amd64, linux/arm64** (BREAKING; `go.mod`, `scripts/build.sh`, `.goreleaser.yaml`, `deployments/Dockerfile`, `deployments/Dockerfile.goreleaser`, `Makefile`, `internal/config/config.go`, `config.yaml`, `chtypes.lock` (new), `scripts/fetch-chtypes.sh` (new), `.github/actions/setup-env/action.yml`, `.github/workflows/ci.yml`): the native type layer above needs cgo for `dlfcn` (no C library linked, no header). cgo is now unconditional: `CGO_ENABLED=0` is gone from every build path, and the `make audit-cgo` target that policed the old no-cgo build has been **removed** along with it (`make binary-analysis` is now `size` + `deadcode`). Because chtypes publishes artifacts only for darwin-arm64, linux-amd64 and linux-arm64, **Windows, FreeBSD and darwin/amd64 builds are discontinued** — `.goreleaser.yaml`'s matrix drops from 8 targets to 3, and the release archives/checksums/GHCR image narrow to match. The runtime image moves from an Alpine/musl builder + `distroless/static` to `golang:1.27-bookworm` (glibc, ships gcc) + `distroless/cc-debian12` (glibc + libstdc++, which the SDK's shared library needs) and bakes the pinned chtypes artifact into the image at `/opt/chtypes/artifacts` via a new `chtypes.lock` (exact file + sha256 per platform/line) and `scripts/fetch-chtypes.sh --frozen` wrapper, so the container has no first-request download. `go.mod` moves to `go 1.27`. Only processes running the `api` role load the artifact, so an ingest-worker-only or sweeper-only process needs none installed; the glibc requirement is 2.34 or later. New boot config: `clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY` lets an operator point at an explicit registry directory instead of the SDK's own search path (the shipped image instead sets the SDK's own `CHTYPES_REGISTRY` env var directly). CI's `unit`/`integration`/`e2e` jobs fetch and cache the pinned artifact (`setup-env`'s new `chtypes` input) and set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` so a missing artifact fails the job instead of silently skipping the chtypes-backed tests. `GOLANGCI_LINT_VERSION` bumped `v2.11.4` → `v2.13.2`: the `go 1.27` bump panics `v2.11.4`'s type checker on every package; `v2.13.0` is the oldest release whose changelog claims go1.27 support, but it panics in this tree for a different reason (`nilness`/`honnef.co/go/tools@v0.8.0-rc.1` crashing while analyzing a third-party dependency), fixed once that dependency moves past its release candidate in `v2.13.1`. The cross-toolchain approach this bullet originally described was replaced before landing — see the release-pipeline entry above. - **Boot refuses an unbound `WH_*` environment variable and an unusable `data_dir`** (BREAKING; `internal/config/check.go` (new, + tests), `internal/config/{config,persistence}.go`, `cmd/wavehouse/main.go`, `docs/src/integrations/diagram-png.mjs`): the environment half of the strict YAML loader. `config.Load` now errors, naming every offender, on a `WH_*` variable that no `Config` field binds — the two variables read outside the struct, `WH_CONFIG` and `WH_LOG_LEVEL`, are exempt — `WH_DEDUPE_ENABLED=true` left in a compose file from before the settings-directory move, or a misspelling, was set, ignored, and believed. **An existing deployment that still exports a variable this release moved to the settings directory stops booting until it is unset**; the upgrade runbook in `deployment.md` gains that audit. Only the `WH_` prefix is checked, since the environment always carries unrelated names; the one outside source that shares it — Kubernetes service-link variables for a Service named `wh` or `wh-*` — is named in the error with the `enableServiceLinks: false` remediation, and the docs build's opt-out knob is renamed from `WH_SKIP_DIAGRAM_PNG` to `DOCS_SKIP_DIAGRAM_PNG` so an exported one no longer refuses a local boot. Right after `Load`, before ClickHouse or the settings directory are touched, `config.CheckDataDir` probes `data_dir` and refuses boot on any of: an empty or blank value (reachable through `WH_DATA_DIR=`), refused outright since the ancestor walk would otherwise fall back to the working directory and NATS and Pebble state would land under it; a path that exists and is not a directory; a dangling symlink at `data_dir` or any component above it (the walk to the nearest existing ancestor uses `Lstat`, so a failed mount is not skipped over as "does not exist" and passed in an unrelated directory); and a directory the process cannot write to — or, when it does not exist, an unwritable nearest ancestor — probed by creating and removing one temp file. So an unusable `data_dir` refuses boot before schema discovery rather than after it; a permission denial — on the probe, or on reaching the path at all through a parent without search permission — carries the UID-65532 remediation (a bind mount owned by root is the typical cause), and that hint string is now shared with `LogStorageInitError`. `EnvConfig` and `EnvLogLevel` join `EnvSettingsDir` as the exported names for the process-level variables. Boot is the validator for the non-hot-reloadable half — there is no dry-run subcommand, by decision on #530: boot config only takes effect through a restart, so the restart is where it is checked, and the docs say so. Closes #530. @@ -81,7 +81,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **A policy `check` on a column the table cannot accept is now refused instead of silently unenforced** (BREAKING; `internal/api/ingest.go`, `internal/discovery/discovery.go`, `docs/src/content/docs/{api.md,access-control.mdx}`): a `check` clause naming a column the table does not have, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one can never be enforced — the published row carries one slot per insertable column, so an auto-injected value for anything outside that set is dropped on the way out, and an ephemeral column is never stored even though the row does carry it. The record inserted **without** the value the policy required and answered `200 {"ok":true}`. Demonstrated on this branch: a `check` of `tenant _eq {{ jwt.tenant }}` against a `MATERIALIZED tenant` column published `columns:["page"], row:["/a"]` — the tenant constraint absent from the row entirely. It is now refused, naming every offending column and the reason, on **every** insert by that role until the policy or the table is corrected: a single-object request answers `403`, while a batch answers `200` with the same message against each record in `results` — the batch is still read to the end and reports per record, as it does for any other rejection. Policy validation cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**; `wavehouse validate` will not tell you. -- **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` / `InsertableColumns` / `InsertableColumnNames`, memoized per table since the ingest path would otherwise rebuild them once per record. `EPHEMERAL` stays insertable — it is insert-only by construction, never stored, confirmed on the same server rather than assumed. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. +- **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` / `InsertableColumns` / `InsertableColumnNames`, memoized per table since the ingest path would otherwise rebuild them once per record. `EPHEMERAL` columns are likewise left out of the envelope, and this entry's earlier claim that they stay insertable is superseded by the type-layer entry above: an `EPHEMERAL` value is accepted as input wherever the format names columns (the JSON family and the `…WithNames` formats) and feeds the `DEFAULT`s that read it, but it is never stored, selected or published, and a positional CSV or TSV body carries the wire columns only. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. - **Ingest reads the request body up front, and the per-record decisions sit behind interfaces** (`internal/api/{ingest,ingest_seams,bufpool,record_reader}.go`, `internal/stream/hub.go`, `internal/ingest/compact.go`): responses are unchanged except at the body cap and one new read-failure body (`400 {"error":"invalid request body"}`, when the body cannot be read at all — a malformed transfer encoding or a truncated upload, which previously surfaced through the decoder as `invalid json`), and at the cap the `413` is now decided before any record is processed: an over-cap batch no longer ingests the prefix it had already decoded, and an over-cap single-object body whose first object was followed by an oversized tail — which used to answer `200` after ingesting that one object — now answers `413`. Both are improvements, since a client retrying a `413` can no longer double-insert a prefix, but they are behavior changes and the memory profile changes too (see below); this is the seam work the native type layer lands against. The handler now reads the whole (already `MaxBytesReader`-capped) body into a pooled `*bytes.Buffer` and runs the record readers over those bytes rather than the live connection — so the `413` surfaces at that read instead of mid-iteration (same status, same message), and the `415` is decided from the header before a single byte is read. Three decision points became interfaces with default implementations that delegate to today's code unchanged: `RecordValidator` (schema validation + timestamp canonicalization — the two calls stay where they are, with the check-clause block between them, since merging them would move checks onto canonicalized values), `InsertChecker` (the `_eq` and `_in` comparisons), and `stream.RowEvaluator` (row visibility, reached by both the live fan-out and replay through the one shared admission step). **The memory profile is not unchanged, and that is the deliberate part.** Streaming meant peak resident bytes on the order of one record: the NDJSON path scanned line by line and the array path let `json.Decoder` compact after each element. Peak is now O(body) per in-flight request — and `bytes.Buffer` grows by doubling, so the peak allocation can exceed the body cap before `MaxBytesReader` errors. `maxPooledBufferBytes` (1 MiB) caps what a request hands *back* to the pool, not its peak, and nothing in `internal/api` bounds total in-flight bytes, so the ceiling is concurrency × the 16 MiB data-plane cap — which has no operator knob (`maxRequestBytes` is test-only), so the outer limit is the reverse proxy's, which the reverse-proxy guide already advises setting. Kept because it is the shape the native type layer lands against, which needs the body addressable rather than consumed; a bound on total in-flight ingest bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). Operators fronting large batches at high concurrency should size for it or cap body size at the proxy. All three are nil-safe: an un-wired handler or `Hub` uses the default rather than panicking past the check. Also new: `ingest.EncodeCompactRow`, which renders a record as one `JSONCompactEachRow` line — inert in this commit, and the encoder every published row went through at that point in the release (superseded: the published row is now ClickHouse's own export, and `EncodeCompactRow`, `RecordValidator` and `InsertChecker` are gone — see the type-layer entry). @@ -93,7 +93,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The landing page's live demo now reads from the stats deployment's new WaveHouse Cloud backend** (`docs/src/components/LiveDemo.astro`, `docs/scripts/screenshot.mjs`): the GitHub-activity dogfood deployment behind the hero panel (Wave-RF/WaveHouse-Stats) moved off its self-managed AWS infrastructure onto WaveHouse Cloud, so `BASE_URL` — the origin `@wavehouse/sdk` queries in the visitor's browser — points at `https://iefrrvavd5akvphk7pq3.wavehouse.app` instead of `https://stats.wavehouse.dev`, ahead of the AWS stack being torn down. The `PUBLIC_WAVEHOUSE_STATS_URL` build-time override is unchanged, so a fork or staging docs build still redirects the panel without a code edit. **`DEMO_HOST` deliberately stays `stats.wavehouse.dev`** — the demo *site* is still served there and is still what the panel's chrome label and "Full demo" link should show; the migration splits the site from the API origin behind it, and the two constants now carry comments saying so. Verified against the new deployment before the switch: all five pipes the panel reads (`gh_summary`, `gh_activity_recent`, `gh_events_per_minute`, and the pre-#19 `gh_stars_total` / `gh_forks_total` fallbacks) return `200` with the same row shapes, the structured-query backfill fallback (`POST /v1/query?table=gh_events`) matches its old-backend response byte for byte, `GET /v1/stream?table=gh_events` opens an SSE stream, and CORS is unchanged (`Access-Control-Allow-Origin: *`, `X-Cache` exposed) so the cross-origin browser reads keep working from the docs site. The new backend is already the live ingest target — it reported more recent events than the old one at cutover (5,175 vs 5,038 over 7d) — which is the other half of why the panel had to follow it. `screenshot.mjs`'s `networkidle` note is retargeted to "the stats demo backend" rather than naming a host it no longer connects to. -- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is ClickHouse's spelling (`"2026-06-21 04:00:00.123"`, in the column's zone, no offset) where it was RFC 3339 UTC, and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification is unchanged — the same `code`/`retryable` table of the query paths — and each tenant's reader connections are capped at its `max_open_conns`. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. +- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. Each tenant's reader connections are capped at its `max_open_conns`, per pool identity (URL, user, database and TLS settings). Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. - **WH001 (no hard-wrapped prose) now applies to every tracked Markdown file, with no carve-out** (`.github/.markdownlint.json` (deleted), `.claude/.markdownlint.json` (deleted), `.claude/skills/integration-astro-view-transitions/` (deleted), `.markdownlint-cli2.jsonc`, `.github/workflows/README.md`, `.claude/skills/pm-triage/references/routine.md`, `AGENTS.md`, `scripts/docs-prose.sh`, `.github/prompts/docs-review.md`, `docs/src/content/docs/claude-code.md`, `docs/src/content/docs/development.md`, `.claude/agents/docs-reviewer.md`): two path-scoped configs had switched WH001 off under `.github/` and `.claude/` ever since [#489](https://github.com/Wave-RF/WaveHouse/pull/489) introduced the rule — baked in from the start rather than added in response to a discovered problem — which left the repo documenting the rule three ways and disagreeing with itself: `CONTRIBUTING.md` promises contributors `make lint` enforces it *everywhere*, while `AGENTS.md` and the `.markdownlint-cli2.jsonc` header wrote up the carve-out. Not theoretical: on [#520](https://github.com/Wave-RF/WaveHouse/pull/520) a reviewer correctly flagged a hard-wrapped bullet in `.github/workflows/README.md`, an agent pointed at `"WH001": false` for that path and pushed back, and the reviewer recorded a *learning* never to flag WH001 there — the wrong invariant, learned off the wrong side of the contradiction ([#521](https://github.com/Wave-RF/WaveHouse/issues/521)). Both configs are deleted — each held nothing but the override, so the root `.markdownlint.json` governs again — and the 51 hard-wrapped paragraphs they were hiding are joined: 41 in `.github/workflows/README.md` and 10 in `.claude/skills/pm-triage/references/routine.md`, mechanical joins with no wording changed and every fenced block, table row, and heading byte-identical either side of the reflow. Deleted with them: the wizard-installed PostHog skill at `.claude/skills/integration-astro-view-transitions/` — 9 files, ~1,456 lines, including an 809-line `EXAMPLE.md` copied wholesale from `PostHog/context-mill`. Its integration job finished in [#277](https://github.com/Wave-RF/WaveHouse/pull/277), nothing in the repo calls it, and the docs-site setup it once described is documented where it belongs — in `docs/src/components/PostHog.astro` and this file. Keeping unowned third-party prose in the tree means content that drifts silently on every upstream bump and that nobody here reviews; it was also the single file that would have needed a special-case lint exclusion, so removing it is what lets WH001 apply with **no exception at all** rather than one documented one. Its two inventory rows in `claude-code.md` go with it, as does the now-dead `docs/posthog-setup-report.md` entry in the `scripts/docs-prose.sh` denylist (the wizard's other artifact, deleted back in [#502](https://github.com/Wave-RF/WaveHouse/pull/502)) and the copies of that denylist in `AGENTS.md` and `.github/prompts/docs-review.md`, which the script's header requires be kept in lockstep. Review of the change then turned up four more things the exclusion had been hiding, all fixed here: **WH001 has a blind spot** — `no-hard-wrapped-prose.mjs` classifies any line indented four or more spaces as an indented code block, so a *nested* list item is never joined, which left three hard-wrapped bullets in `.github/workflows/README.md` §"Adding a job" that the autofix could not see (unwrapped by hand; they were the last hard-wrapped prose paragraphs in the repo) and made `AGENTS.md`'s and `development.md`'s "a list item is joined as a unit" wrong for nested items (both now state the four-space caveat); the `scripts/docs-prose.sh` header told readers to keep its denylist in lockstep with **two** sibling copies when there are **three** — the missed one being `.claude/agents/docs-reviewer.md`, the gating subagent's own system prompt, which had in fact been silently out of sync for the whole life of the `posthog-setup-report.md` exclusion; the `.markdownlint-cli2.jsonc` header's "applies to every tracked Markdown file" was exact for WH001 but not WH002, which returns early on anything that isn't `.mdx`; and the job-graph diagram omitted `docs-deploy`'s `needs` edges from `unit`, `integration`, and `e2e`, contradicting invariant 2 three lines below it. The denylist also drops its `PERF-CLAIMS-REVIEW.md` entry — unlike the wizard artifact this one names a file that was **never tracked** at all, so it guarded a hypothetical; the list's other general cases are patterns (`*.draft.md`, `*.old.md`) that already cover a one-off review document, and a literal filename restated in four places is the outlier. `scripts/docs-prose.sh all` still resolves the same 27-file prose set. @@ -101,7 +101,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The Go-side validation, timestamp canonicalization, row-filter evaluation and compact encoding** (`internal/discovery/{validation,timestamp}.go`, `internal/policy/{rowfilter,numeric}.go`, `internal/ingest/compact.go`, `internal/api/ingest_seams.go`, `internal/stream/hub.go`): `discovery.Validate` and `CanonicalizeTimestamps`, `policy.RowVisible`, `ingest.EncodeCompactRow`, the `RecordValidator`/`InsertChecker` seams and `stream.NumericSpecOf` are gone. Their jobs — parsing a record, rendering a timestamp, deciding a row filter, writing the positional row — are ClickHouse's own now, through chtypes (see the type-layer entry in Changed). `typelayer.InsertSettings()` is the one static set of parsing settings the worker's `INSERT` and the ingest parse share. -- **`policy.LiteralValue` and `policy.CanonicalNumericLiteral`** (`internal/policy/{canonical,policy}.go`): the marker type and the numeric re-reading of a policy-authored insert-check literal. Insert checks are chtypes filters now, so a literal binds as written and ClickHouse reads it under the column's type — there is no second, numeric reading at compare time. A `_eq: "1.0"` against a `UInt64` used to admit a stored `1`; it is now ClickHouse's code 53 `TYPE_MISMATCH` per row (`422`), and the fix is to write a literal the column can read. `CanonicalScalar` stays: it is still the one rendering layer for a JWT claim. The released-version entry further down this file describing `LiteralValue` as shipped behaviour is left as history. +- **`policy.LiteralValue` and `policy.CanonicalNumericLiteral`** (`internal/policy/{canonical,policy}.go`): the marker type and the numeric re-reading of a policy-authored insert-check literal. Insert checks are chtypes filters now, so a literal binds as written and ClickHouse reads it under the column's type — there is no second, numeric reading at compare time. A `_eq: "1.0"` against a `UInt64` used to admit a stored `1`; the literal is no longer re-read numerically, so an insert check on it matches no row and the record is refused with `403` (the strict integer cast treats a non-canonical value as matching nothing), and the fix is to write a literal the column can read. `CanonicalScalar` stays: it is still the one rendering layer for a JWT claim. The released-version entry further down this file describing `LiteralValue` as shipped behaviour is left as history. - **The policy's entire HTTP surface — `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, and the SDK's `wh.policy` namespace** (`internal/api/policy.go` + `policy_test.go` (deleted), `internal/api/{router,router_test}.go`, `cmd/wavehouse/main.go`, `clients/ts/src/policy.ts` (deleted), `clients/ts/src/{client,types,index}.ts`, `tests/e2e/sdk/{admin,query,ingest,streaming}.test.ts` + `settings.ts`, `docs/src/content/docs/{api.md,access-control.mdx,settings-directory.mdx,architecture.md,configuration.mdx,development.md,reverse-proxy.mdx,sdk/admin.md,sdk/reference.md}`, `AGENTS.md`; closes [#514](https://github.com/Wave-RF/WaveHouse/issues/514)): both endpoints were born alongside `PUT /v1/ops/policy` and outlived it when [#508](https://github.com/Wave-RF/WaveHouse/pull/508) deleted the policy write API; with files as the only write path, the policy is read, edited, and validated where it lives, so the whole read/dry-run surface goes too. The dry run had also kept its original lenient decoder while adoption became strict, certifying `{"valid": true}` for documents a reload would refuse — a misspelled operator key (`"eq"` for `"_eq"`) silently dropped into a filter that disables row security, the exact fail-open [#460](https://github.com/Wave-RF/WaveHouse/issues/460) demonstrated; deleting it removes the last non-strict policy decode site, closing #514 (the other five sites it cites were deleted or made strict by #508). Its replacement is `wavehouse validate`, which enforces strictly more (the cross-file role references against `roles.json` were invisible to a single-document dry run). The break-glass story narrows accordingly: the operator key can still trigger `POST /v1/ops/settings/reload`, whose findings report exactly why a rejected directory was refused — what it can no longer do is read back the adopted snapshot over HTTP; a bad edit still never breaks a running server (the previous good snapshot stays adopted). The e2e suite's read-modify-write helper pattern moves from `wh.policy.get()` to reading the harness-owned `policies.json` directly (`readPolicyFile()` — the file *is* the adopted policy there, since `setPolicy` fails unless the reload reports adoption). The policy document types (`Policy`, `TablePolicy`, `RolePermissions`) stay exported from the SDK — they describe `policies.json` and the e2e harness consumes them; `GET /v1/ops/pipes[/{name}]` is untouched. diff --git a/README.md b/README.md index 698c7aee..2edbc80b 100644 --- a/README.md +++ b/README.md @@ -121,19 +121,21 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:dev \ Swap in `:vX.Y.Z` and `release.yml` for a release image. Pin the signer either way. `--repo` alone accepts an attestation from any workflow in the repo. +The published images bake the chtypes artifact for ClickHouse 26.8 only. Against any other ClickHouse line, bind-mount a directory holding that line's artifact and set `WH_CHTYPES_REGISTRY` to it (see [chtypes artifacts](https://wavehouse.dev/deployment#chtypes-artifacts)). + ### C. `go install` (binary, no Docker) ```bash go install github.com/Wave-RF/WaveHouse/cmd/wavehouse@latest ``` -`go install` compiles from source with cgo enabled (requires a C toolchain and glibc — Linux amd64/arm64 or macOS arm64) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse loads at start. Fetch it once before the first run: +`go install` compiles from source with cgo enabled (requires a C toolchain, and on Linux glibc 2.34 or later — Linux amd64/arm64 or macOS arm64) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse loads at start. Fetch it once before the first run: ```bash go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch ``` -This downloads 160–290 MB into the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision); point `WH_CHTYPES_REGISTRY` elsewhere if you keep it somewhere else. +(From a checkout, `scripts/fetch-chtypes.sh` fetches the build pinned in `chtypes.lock`.) This downloads 160–290 MB into the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision); point `WH_CHTYPES_REGISTRY` elsewhere if you keep it somewhere else. ```bash wavehouse bootstrap ./settings # starter settings directory, every key at its default diff --git a/SECURITY.md b/SECURITY.md index 44bd6426..694795e7 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -26,9 +26,9 @@ WaveHouse handles data and enforces strict isolation: - **JWT validation**: The JWT middleware always runs (there is no on/off switch). Signing supports either an HMAC shared secret or a remote JWKS endpoint (`auth.jwks_url`, one verifier per tenant over a nested settings directory, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set — the HMAC secret is boot config shared by every tenant; a JWKS response is capped at 1 MiB). Accepted signing algorithms are restricted to the configured verifier's family — HMAC accepts only `HS256/384/512`, JWKS only the asymmetric set (`RS*`/`ES*`/`PS*`/`EdDSA`) — and the token's `alg` header is validated before any key material is used, so `alg: none` and algorithm-confusion attacks (re-signing with `HS256` against a JWKS deployment's public key) are rejected. A request with no token, or an invalid/expired one, falls back to the policy `default_role`; elevated access requires a valid token, and a denied request that carried a bad token fails loud (`401`) rather than as a bare `403`. With neither a secret nor a JWKS URL configured, no token validates at all: the verifier refuses rather than hand the JWT library an empty HMAC key, which it would accept as a match for a token signed with one. The one token outcome that never falls back is a tenant whose JWKS has not been fetched yet: its token-bearing requests are refused with `503` + `Retry-After` rather than evaluated under a lesser role, so a token is never silently downgraded while the keys that would verify it are still on their way. - **Role-based access control**: Roles are extracted from a configurable JWT claim path. Non-admin roles have per-table, per-column policies enforced on ingest, query, and the live SSE stream; row-level rules split by path — insert `check` constraints are enforced (and auto-injected) on ingest, while select `filter` predicates apply to structured queries and the live SSE stream (the stream's in-memory row-filter comparison has a documented fail-closed boundary — see the [access-control docs](https://wavehouse.dev/access-control#where-each-rule-is-enforced)); the admin role (`policy.admin_role`, `"admin"` by default, exact case-sensitive match) bypasses them. The configured non-JWT operator key (`auth.operator_key`) likewise bypasses per-role policy — a matching request is authorized as a full-access platform operator without a JWT; treat it as an admin secret. A request presenting a *non-matching* operator key is logged at `WARN` and counted by `wavehouse_auth_operator_key_failures_total`, so probing of that credential is observable and alertable. -- **Input validation**: JSON payloads are validated against ClickHouse schemas before processing. +- **Input validation**: ingest bodies (JSON, NDJSON, CSV and TSV) are validated against ClickHouse schemas before processing, by ClickHouse's own parser running in-process as a native shared library (the chtypes artifact) loaded by the API process. Every untrusted ingest body therefore reaches native code, which is why a request body is capped at 16 MiB and why the artifact's integrity is pinned (below). - **Query passthrough**: Raw SQL via `POST /v1/ops/query` is restricted to the admin role — the same `RequireAdmin` gate as the rest of `/v1/ops/*`. A request with no/invalid token resolves to the `default_role`, which in a production config is not the admin role (setting `default_role` equal to the admin role is a loudly-warned, dev-only escape hatch), so it cannot reach this endpoint — the one exception is a request presenting the configured `auth.operator_key`, which reaches the whole `/v1/ops/*` surface (including this endpoint) without a JWT and even under a deleted policy, so treat that key as an admin secret. Raw SQL has no per-statement scope check (a full SQL parser would be needed to authorize predicates), so the role gate is the entire authorization story. Non-admin callers use structured queries (`POST /v1/query?table={table}`, validated against schema with permission injection) or named pipes (`GET/POST /v1/pipes/{name}`); raw-SQL grants to non-admin roles via the policy engine are no longer supported (the `raw_sql` field on policies has been removed). -- **Supply chain**: Third-party GitHub Actions are pinned to full commit SHAs (enforced by the repository's Actions settings — `sha_pinning_required`). `govulncheck` runs on every push/PR. Dependabot opens weekly grouped PRs for Go modules, GitHub Actions, and the npm packages — one grouped PR covering the docs site, TS SDK, and E2E tests via the root pnpm workspace. Released artifacts ship signed [Sigstore](https://www.sigstore.dev/) build-provenance attestations — verify the container image with `gh attestation verify oci://ghcr.io/wave-rf/wavehouse: --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml` (the rolling `:dev` image is signed by `publish-dev.yml` — swap the `--signer-workflow`; `--repo` alone accepts an attestation from any workflow in the repo), a downloaded release-binary archive with `gh attestation verify --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml`, and the `@wavehouse/sdk` package via its npm provenance badge or `npm audit signatures`. (Provenance covers the published binaries and image, not `go install`, which compiles from source.) +- **Supply chain**: The native parser library is pinned in `chtypes.lock` by exact file and sha256 per platform and ClickHouse line, fetched with `scripts/fetch-chtypes.sh --frozen` (which refuses a file whose hash differs), and baked into the published images, so a container makes no download at runtime. Third-party GitHub Actions are pinned to full commit SHAs (enforced by the repository's Actions settings — `sha_pinning_required`). `govulncheck` runs on every push/PR. Dependabot opens weekly grouped PRs for Go modules, GitHub Actions, and the npm packages — one grouped PR covering the docs site, TS SDK, and E2E tests via the root pnpm workspace. Released artifacts ship signed [Sigstore](https://www.sigstore.dev/) build-provenance attestations — verify the container image with `gh attestation verify oci://ghcr.io/wave-rf/wavehouse: --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml` (the rolling `:dev` image is signed by `publish-dev.yml` — swap the `--signer-workflow`; `--repo` alone accepts an attestation from any workflow in the repo), a downloaded release-binary archive with `gh attestation verify --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml`, and the `@wavehouse/sdk` package via its npm provenance badge or `npm audit signatures`. (Provenance covers the published binaries and image, not `go install`, which compiles from source.) ## Disclosure Policy diff --git a/deployments/Dockerfile b/deployments/Dockerfile index d4b5c6ae..7b35b2c8 100644 --- a/deployments/Dockerfile +++ b/deployments/Dockerfile @@ -33,7 +33,7 @@ RUN --mount=type=cache,target=/go/pkg/mod \ # Bake the pinned chtypes artifact(s) into the image so the container has no # runtime network dependency and no first-request download stall. chtypes.lock -# is this repo's pin (schema-1, exact file + sha256 per platform/line); +# is this repo's pin (schema 2, exact file + sha256 per platform/line); # scripts/fetch-chtypes.sh wraps the SDK's own CLI with --frozen, which # refuses anything the lock doesn't name. `go run @` # resolves and builds the SDK's CLI straight from its module proxy — it diff --git a/deployments/compose/standalone.yaml b/deployments/compose/standalone.yaml index bc3d1500..48820d21 100644 --- a/deployments/compose/standalone.yaml +++ b/deployments/compose/standalone.yaml @@ -25,8 +25,9 @@ services: WH_DATA_DIR: /app/data # ClickHouse wiring (addr, ports, database, user, tls, headers, pool # sizes) is in ./settings/config.json — the settings directory below — - # where it hot-reloads. Only the password (a secret) and the connection - # ceiling are env; the bundled ClickHouse has no password. + # where it hot-reloads. Only the password (a secret), the connection + # ceiling and the chtypes artifact directory (below) are env; the + # bundled ClickHouse has no password. # WH_CH_PASSWORD: "" # WH_CH_MAX_TOTAL_CONNS: "0" # deployments/Dockerfile bakes the pinned chtypes artifacts diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 581447ba..4e401791 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -190,9 +190,9 @@ The rules, in order: 2. **An empty (or `["*"]`) `allow_columns` means "all columns"** — every column not in `deny_columns` is permitted. Use this with `deny_columns` for a blocklist posture: see everything *except* a few sensitive columns. 3. **A non-empty `allow_columns` is an allowlist** — only the named columns (and never the denied ones) are permitted. -On a structured query (`POST /v1/query?table={table}`) the allowlist is a **hard cap on every column the query references — in any clause**: the projection, an aggregation argument, `filters`, `group_by`, `order_by`, and `time_range`. Naming a disallowed column anywhere is rejected with `403 column "x" not allowed`. A full-row read is requested explicitly with `"select_all": true`, which expands to exactly the columns the role may read — never a raw `SELECT *` that could include a denied column; if the role is allowed *no* columns, the read is rejected (`403`) rather than returning empty rows. **Omitting `columns` (or sending `[]` / `""`) returns nothing** — a request for no data — so a hidden column can't leak by being left out, grouped on, or filtered on to infer its values. (Note: in a query, `["*"]` is the *literal column named `*`*, not a wildcard — use `select_all` for all columns. In `allow_columns`, `["*"]` is still the all-columns wildcard.) On insert (`POST /v1/ingest?table={table}`) the rule is enforced by compiling the role its *own* copy of the table schema, without the columns it may not write — so a record naming one is refused by ClickHouse's parser with code **117**, `Unknown field found while parsing JSONEachRow format: x`, the same answer a column the table does not have gets. The message no longer confirms whether the column exists. On live streams, denied columns are silently **stripped** from each event rather than rejecting the connection. The structured-query and live-stream paths defer to the **same** per-column decision (`IsColumnAllowed`), so the two read surfaces enforce identical column visibility and can't drift apart. +On a structured query (`POST /v1/query?table={table}`) the allowlist is a **hard cap on every column the query references — in any clause**: the projection, an aggregation argument, `filters`, `group_by`, `order_by`, and `time_range`. Naming a disallowed column anywhere is rejected with `403 column "x" not allowed`. A full-row read is requested explicitly with `"select_all": true`, which expands to exactly the columns the role may read — never a raw `SELECT *` that could include a denied column; if the role is allowed *no* columns, the read is rejected (`403`) rather than returning empty rows. **Omitting `columns` (or sending `[]` / `""`) returns nothing** — a request for no data — so a hidden column can't leak by being left out, grouped on, or filtered on to infer its values. (Note: in a query, `["*"]` is the *literal column named `*`*, not a wildcard — use `select_all` for all columns. In `allow_columns`, `["*"]` is still the all-columns wildcard.) On live streams, denied columns are silently **stripped** from each event rather than rejecting the connection. The structured-query and live-stream paths defer to the **same** per-column decision (`IsColumnAllowed`), so the two read surfaces enforce identical column visibility and can't drift apart. -On ingest (`POST /v1/ingest?table={table}`) the same lists cap what a role may write, and a column it may not write is indistinguishable from one the table does not have: the record is refused by ClickHouse's own parser with code `117` (`Unknown field found while parsing …`), a `400` — per record in a batch, or for the whole request when the body is a single object or has a `header=present` header naming it. There is no separate `403` for a denied column on insert, and the published row never carries one. +On ingest (`POST /v1/ingest?table={table}`) the same lists cap what a role may write, and a column it may not write is indistinguishable from one the table does not have: the record is refused by ClickHouse's own parser with code `117` (`Unknown field found while parsing …`), a `400` — per record in a batch, or for the whole request when the body is a single object or has a `header=present` header naming it. The role is compiled its own copy of the table schema, without the columns it may not write, so the message does not confirm whether the column exists. There is no separate `403` for a denied column on insert, and the published row never carries one. ## Row-level security @@ -367,7 +367,7 @@ The same policy drives every data path, but not every field is meaningful on eve :::caution[Live streams enforce column and row policy, but not resource limits] SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, ClickHouse's code 53; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine cannot compile (a constant the column type cannot read) withholds every row for that role until the filter or the table changes — logged once, not per event. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, ClickHouse's code 53; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine cannot compile (a constant the column type cannot read) withholds every row for that role until the filter or the table changes — logged once, not per event. Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 5908ff9d..7293fc52 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -90,7 +90,8 @@ When ClickHouse fails a query on [`POST /v1/query`](#post-v1querytabletable--str | 400 | `clickhouse.rejected` | `false` | ClickHouse read the statement and refused it: bad SQL, an unknown table, column or identifier, a type mismatch — any exception code not listed below; also a `413` with no exception code from a proxy in front of ClickHouse (the statement is too large for it). Sent again unchanged, it fails the same way | | 400 | `clickhouse.limit_exceeded` | `false` | The query outran a limit it ran under: rows read or returned, bytes (`TOO_MANY_ROWS`, `TOO_MANY_BYTES`, `TOO_MANY_ROWS_OR_BYTES`), or, on `/v1/query`, the role's own `max_memory_usage` cap or a `max_execution_time` cap no longer than the tenant's `clickhouse.query_timeout` (`TIMEOUT_EXCEEDED`, `TOO_SLOW`, `MEMORY_LIMIT_EXCEEDED`). Narrow the query | | 403 | `clickhouse.access_denied` | `false` | The ClickHouse user WaveHouse connects as lacks a grant the statement needs (`ACCESS_DENIED`). Grant it, or run something it may. Logged at `WARN` too | -| 502 | `clickhouse.misconfigured` | `false` | ClickHouse refused the credentials or database WaveHouse connects with: a wrong password, an unknown or expired user, a refused address, the database denied, a `401`/`403` from a proxy in front of it, or any other `3xx`/`4xx` with no exception code — a wrong path, a redirect WaveHouse does not follow (a codeless `408`/`429` is `503`, a `413` is `400 clickhouse.rejected`). Every query fails until the operator fixes the tenant's `clickhouse` settings or `WH_CH_PASSWORD`, so retrying does not help. Logged at `WARN` | +| 502 | `clickhouse.misconfigured` | `false` | ClickHouse refused the credentials or database WaveHouse connects with: a wrong password, an unknown or expired user, a refused address, the database denied, a `401`/`403` from a proxy in front of it, or any other `3xx`/`4xx` with no exception code — a wrong path, a redirect WaveHouse does not follow, or a read the server refused as a write (`READONLY`, from a `readonly=1` profile or a statement the mutation classifier missed) (a codeless `408`/`429` is `503`, a `413` is `400 clickhouse.rejected`). Every query fails until the operator fixes the tenant's `clickhouse` settings or `WH_CH_PASSWORD`, so retrying does not help. Logged at `WARN` | +| 502 | `clickhouse.response_too_large` | `false` | The response outgrew the 64 MiB the reader buffers (`/v1/query`, pipes and `/v1/ops/query` alike). Narrow the query or add a `LIMIT` | | 503 | `clickhouse.unavailable` | `true` | ClickHouse, or the way to it, could not take the query now: connection refused or dropped, a timeout, too many queries, memory pressure, lost replicas or Keeper, or a `502`/`503`/`504`/`429`/`408` from a proxy. `Retry-After: 5` | | 500 (`/v1/query`, pipes) / 502 (`/v1/ops/query`) | `clickhouse.unknown` | `true` | A failure with no verdict: no exception code and no recognizable transport error | @@ -222,7 +223,7 @@ Every other route answers `404`, including every tenant route. Under `/v1/ops`, Validates a body of records against the ClickHouse schema for `{table}` and publishes each accepted one to the message queue. Returns immediately — ClickHouse insertion happens asynchronously via the batch consumer. A single-object body answers `{"ok":true}` (or `{"duplicate":true}` when dedup is on); every other body answers the [batch summary](#batch-ingest). -**The body goes to ClickHouse's own parser as-is.** WaveHouse never decodes a record: validation, type coercion, `DEFAULT` substitution and timestamp parsing are ClickHouse's own, running in-process via [chtypes](/deployment#chtypes-artifacts) (`internal/typelayer`) — the exact code path a real `INSERT` runs. A rejection therefore carries ClickHouse's own message and its numeric error number as `exception_code` (the string `code` stays the failure class, as on the query paths) rather than a WaveHouse-authored sentence, and there is no separate coercion table to keep in sync with the server. +**The body goes to ClickHouse's own parser as-is.** WaveHouse never decodes a record: validation, type coercion, `DEFAULT` substitution and timestamp parsing are ClickHouse's own, running in-process via [chtypes](/deployment#chtypes-artifacts) (`internal/typelayer`) — the exact code path a real `INSERT` runs. A rejection therefore carries ClickHouse's own message and its numeric error number as `exception_code` rather than a WaveHouse-authored sentence, and there is no separate coercion table to keep in sync with the server. A rejected single record answers `{"error","exception_code"}` with no string `code`; only a refusal of the whole request (a `header=present` header naming a column the table lacks) carries both, `code: "clickhouse.rejected"` and `exception_code`. The SDK reports either as `HTTP_400`, with the number in `error.details`. **`Content-Type` is required and authoritative**: it declares the format and the bytes never override it. @@ -253,7 +254,7 @@ The policy engine authorizes mutations by inspecting the columns being written. **What ClickHouse decides, and what WaveHouse decides.** Everything about a *value* is ClickHouse's: -- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED`, `ALIAS` and `EPHEMERAL` columns — none of the three is ever part of a published row. +- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED` and `ALIAS` columns — neither is ever part of a published row. An `EPHEMERAL` column is accepted as input wherever the format names its columns (the JSON family and the `…WithNames` formats) and feeds the `DEFAULT` expressions that read it, but it is never stored, selected or published; a positional CSV or TSV body carries the wire columns only. - An omitted column, or an explicit `null` on one (WaveHouse pins `input_format_null_as_default`), takes its `DEFAULT` expression — evaluated by ClickHouse, including a volatile one like `now()` — or the type's implicit zero where none is declared, exactly as an `INSERT` naming fewer columns does. - A coercion ClickHouse would make it makes here (a numeric string into an `Int*`, `"true"` into a `Bool`, an out-of-range integer wrapping); anything it would refuse fails synchronously in the ingest response with its real code, rather than surfacing later in the DLQ. `Nullable()` and `LowCardinality()` wrappers are transparent. @@ -263,9 +264,9 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | Status | Body | Cause | | ------ | ---- | ----- | -| 400 | `{"error":"","exception_code":}` | **Per-record.** ClickHouse's parser refused the record; `exception_code` and the message are its own. `117` is an unknown field — which now includes a column the role may not write, and any `MATERIALIZED`/`ALIAS`/`EPHEMERAL` column; `27`/`26` are unparseable input; `6` out of range | +| 400 | `{"error":"","exception_code":}` | **Per-record.** ClickHouse's parser refused the record; `exception_code` and the message are its own. `117` is an unknown field — which now includes a column the role may not write and any `MATERIALIZED`/`ALIAS` column; `27`/`26` are unparseable input; `6` out of range | | 400 | `{"error":"Unknown field found in format header: 'x' at position 1 …","code":"clickhouse.rejected","exception_code":117}` | A `header=present` body whose header names a column the table — or the role's writable set — does not have, or names one twice. ClickHouse refuses the body before reading any record, so the whole request fails and nothing is published | -| 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is `invalid json` | +| 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26 or 27) | | 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | | 400 | `{"error":"missing dedupe id field \"event_id\""}` | **Per-record.** Only with `dedupe.require_id: true`, when the row carries no value for the configured `id_field` (an absent column, a `null` cell or an empty string — the value an omitted `String` id column stores). With `require_id: false` (the default) the row is published un-deduped instead. Either way it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` | @@ -278,6 +279,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | | 422 | `{"error":"validation engine declined: "}` | **Per-record.** chtypes could not evaluate the record at all — the artifact declined the shape, rather than the data being wrong. A `check` clause that could not be evaluated lands here too (`validation engine declined: the insert check for column "x" could not be evaluated`) | +| 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither the record's fault nor an unavailable tenant; logged. Nothing was published | | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now: it is not open (for example, it failed to open on a reload), or a DynamoDB table is throttling, timing out or unreachable; `Retry-After: 5`. Nothing was published, so the retry is safe | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet (its ClickHouse unreachable, or [no pool for it](/settings-directory#clickhouse)), so whether the table exists is not known; `Retry-After: 5`. Decided before the body is read | @@ -301,9 +303,9 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling ClickHouse's own parser accepts under `date_time_input_format=best_effort` (the setting WaveHouse pins, both at chtypes' ingest compile and on the worker's `INSERT`) is accepted — RFC 3339 with any offset, a zone-less `YYYY-MM-DD[ T]HH:MM:SS[.fff]` read in the column's declared zone else the server's default, a Unix-seconds string, a bare integer (ClickHouse 26.8 reads a bare number in a `DateTime64` column as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — send milliseconds as a quoted string, or as a decimal number of seconds), among the other forms its lenient parser reads. It is ClickHouse's grammar, not a reimplementation of it, so whatever a real `INSERT` into this table would accept, ingest accepts, with the same coercions and the same refusals. -**Outbound**, `DateTime`/`DateTime64` values in the NATS/SSE wire row and in `/v1/query` / `/v1/pipes/{name}` results are the exact bytes ClickHouse's writer produces, in the column's declared zone else the server's default: `"2026-06-21 04:00:00.123"`, space-separated, no `Z` suffix, never RFC 3339. Every consumer renders from the same stored value the same way, so SSE and `/v1/query` agree on spelling for a given row **by construction**, with no WaveHouse rewriting step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). The raw-SQL proxy `/v1/ops/query` is the exception: it sets `date_time_output_format=iso`, which keeps trailing fraction zeros and an ISO-8601 `Z`, and is not expected to match the other two byte-for-byte. +**Outbound**, `DateTime`/`DateTime64` values in the NATS/SSE wire row and in `/v1/query` / `/v1/pipes/{name}` results are the exact bytes ClickHouse's writer produces, as RFC 3339 in UTC, whatever zone the column declares (ClickHouse's `date_time_output_format=iso`): `"2026-06-21T04:00:00.123Z"` for a `DateTime64(3)`, `"2026-06-21T04:00:00Z"` for a `DateTime`, with the fraction at the column's own precision, trailing zeros kept. Every consumer renders from the same stored value the same way, so SSE, `/v1/query` and `/v1/pipes/{name}` agree on spelling for a given row **by construction**, with no WaveHouse rewriting step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)), and the strings order the same way the instants do. The raw-SQL proxy `/v1/ops/query` sets the same `date_time_output_format=iso`, so it spells a `DateTime`/`DateTime64` the same way; its other types are returned as ClickHouse renders them under `FORMAT JSON`. A `Date` column is `"2026-06-21"` throughout. -**One server time zone per ClickHouse line, per process.** The in-process parser takes its zone once, when a process first opens a ClickHouse line, and keeps it for the process's lifetime. A tenant whose server reports a different zone than the one that line was opened with is refused on its own — ingest answers `503`, its stream rows are withheld with reason `unavailable` — while every other tenant keeps working. Run tenants whose servers use different zones in separate processes. +**One server time zone per ClickHouse line, per process.** The in-process parser takes its zone once, when a process first opens a ClickHouse line, and keeps it for the process's lifetime. A tenant whose server reports a different zone than the one that line was opened with is refused on its own — ingest answers `503`, a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working. Run tenants whose servers use different zones in separate processes. **Row-level security compares instants, not spellings.** A stream row filter on a `DateTime`/`DateTime64` column is compiled and evaluated by the same engine that validates ingest ([internal/typelayer](/access-control#where-each-rule-is-enforced)), so a filter constant in any spelling ClickHouse would accept in a `WHERE` clause matches the stored instant however the payload spelled it, and a predicate the engine cannot compile withholds every row for that role. @@ -317,7 +319,7 @@ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling Cl | `text/csv; header=absent` | `CSV`, strictly positional: header detection is off (`input_format_csv_detect_header=0`), so every line is a record | | `text/csv` (no `header` parameter) | ClickHouse's default `CSV`: header auto-detection stays on | -`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. With no parameter, the positional fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three computed kinds and that is the field order. +`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. With no parameter, the positional fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column. | Body | Outcome | | --- | --- | @@ -391,20 +393,21 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ | `succeeded` | records validated and published | | `failed` | records rejected — see `results` | | `duplicates` | records skipped by dedup (when enabled) | -| `results` | per-record outcomes, each `{ index, ok\|duplicate\|error }` with `index` the 1-based record position. Truncated to the first 10,000 entries for very large batches (the counts stay authoritative). | +| `results` | per-record outcomes, each `{ index, ok\|duplicate\|error }` with `index` the 1-based record position, plus `exception_code` when the rejection was ClickHouse's parser's. Truncated to the first 10,000 entries for very large batches (the counts stay authoritative). | A `200` is returned whenever the body was read and the records were processed — **even if every record failed**, so branch on `failed`/`results`, not the status code. Per-record problems (a malformed NDJSON line, a non-object array element, a ClickHouse parser rejection, a denied column or a failed `check`) are reported in `results` and the batch continues. Whole-request conditions abort with a non-`200` instead: | Status | Body | Cause | | ------ | ---- | ----- | | 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | -| 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is `invalid json` | +| 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26 or 27) | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (same auth gate as the single-object path; surfaces the token reason) | | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table (checked once, before any record) | | 403 | `{"error":"insert permissions were not resolved for this request"}` | The grant that resolved for the request is not an insert grant; see the [single-record table](#post-v1ingesttabletable--ingest-data). Fails the whole request, an empty array included | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | +| 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither a record's fault nor an unavailable tenant; logged. Nothing was published | | 500 | `{"error":"publish failed"}` / `{"error":"dedupe failed"}` | Message-queue or dedup-backend failure mid-batch, other than a full queue or an unreachable broker (below). After a publish failure the records before it keep their ids, so a whole-batch retry reports those as duplicates; the failing record's id is left to lapse as on the single-object path, and the rest of its window's ids are given back | | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure) or not open, mid-batch; includes `Retry-After: 30`. The records before the refused one keep their ids, and its id and the rest of its window's are given back | | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`), mid-batch. Only under [`mq.backend: nats`](/deployment#external-nats); the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the failing record's id is left to lapse rather than given back — so `Retry-After` is that record's dedupe lease, rounded up to whole seconds, when it was deduped; a record published un-deduped has no lapsing claim to wait out, so `Retry-After: 5` | @@ -421,7 +424,7 @@ A batch aborted partway — a `503` or `500` after some leading records were alr ### `POST /v1/ops/query` — Query ClickHouse -Executes a SQL statement directly against ClickHouse. **WaveHouse proxies the SQL string verbatim to ClickHouse's HTTP interface** — any statement ClickHouse accepts works, including arbitrary DDL/DML/SYSTEM verbs and inline FORMAT directives. Multi-statement input (`SELECT 1; TRUNCATE t`) also works on recent ClickHouse versions where multi-query is enabled by default; older or restrictively-configured servers may reject the second statement with a clear error. Read queries return a JSON array of result rows; mutations/DDL return HTTP 200 with `[]` on success. DateTime columns are ISO-8601 formatted via the upstream `date_time_output_format=iso` setting — server-side rendering that keeps trailing fraction zeros, so a `DateTime64(3)` whole-second value returns `.000Z` here where `/v1/query` renders plain `Z`; other types are returned as ClickHouse renders them under `FORMAT JSON`. +Executes a SQL statement directly against ClickHouse. **WaveHouse proxies the SQL string verbatim to ClickHouse's HTTP interface** — any statement ClickHouse accepts works, including arbitrary DDL/DML/SYSTEM verbs and inline FORMAT directives. Multi-statement input (`SELECT 1; TRUNCATE t`) also works on recent ClickHouse versions where multi-query is enabled by default; older or restrictively-configured servers may reject the second statement with a clear error. Read queries return a JSON array of result rows; mutations/DDL return HTTP 200 with `[]` on success. DateTime columns are ISO-8601 formatted via the upstream `date_time_output_format=iso` setting — the same server-side rendering `/v1/query`, pipes and the stream use, so a `DateTime64(3)` whole-second value returns `.000Z` on all of them; other types are returned as ClickHouse renders them under `FORMAT JSON`. :::note[Inline `FORMAT` overrides the JSON envelope] ClickHouse's inline `FORMAT` clause (e.g. `SELECT 1 FORMAT CSV` or `… FORMAT Pretty`) takes precedence over the URL-level `default_format=JSON` setting. When the SQL contains an explicit `FORMAT`, the proxy forwards ClickHouse's raw response body (CSV, Pretty, TSV, …) and passes through the upstream `Content-Type` header — `text/csv`, `text/tab-separated-values`, etc. — so consumers see the right MIME type. The "extract the `data` array" behavior only applies when ClickHouse returned the `FORMAT JSON` envelope, which is the default. @@ -559,7 +562,7 @@ JSON array of result rows, **rendered by ClickHouse**: the query runs over its H | ClickHouse type | JSON | | --- | --- | -| `DateTime`, `DateTime64` | `"2026-06-21 04:00:00.123"` — space-separated, no `Z`, in the column's declared zone else the server's; byte-identical to the [SSE stream](#get-v1stream--server-sent-events-stream) for the same row (see [Timestamp rendering](#timestamp-rendering)) | +| `DateTime`, `DateTime64` | `"2026-06-21T04:00:00.123Z"` — RFC 3339 in UTC whatever zone the column declares; byte-identical to the [SSE stream](#get-v1stream--server-sent-events-stream) for the same row (see [Timestamp rendering](#timestamp-rendering)) | | `Decimal*` | a JSON **number** (`12.5`), not a string | | `Int64`/`UInt64` past 2^53 | an unquoted number — still lossy in a JavaScript `number`; read it as text if you need every digit | | `FixedString(n)` | a string padded to `n` bytes with `\u0000` | @@ -579,13 +582,12 @@ The inbound request body is capped at 1 MiB; a body over the cap is rejected wit | 400 | `{"error":"unknown column: x"}` | Schema validation error — an unknown column, a bad aggregation, or an unparseable `time_range` `since`/`until` (neither a relative duration nor an RFC3339 timestamp) | | 400 | `{"error":"filter value must not be null"}` | A filter carries `"value": null`. It is refused rather than answered: `col = NULL` is never true, and an empty parameter would silently ask a different question | | 400 | `{"error":"filter value too large: …"}` / `{"error":"query too large: …"}` | A scalar filter value over the 128 KiB ClickHouse's HTTP interface takes for one value once URL-encoded, all of them over the request line, or a query with an `in` list whose SQL is over the 128 KiB form field it travels in. An `in` list itself is never refused for size | -| 500 | `{"error":"clickhouse query: Code: 158. DB::Exception: … (TOO_MANY_ROWS) …"}` | ClickHouse refused the query — a resource limit, a type mismatch, anything else its engine raises. The body carries ClickHouse's own wording | | 403 | `{"error":"forbidden"}` | Role lacks select permission on table | | 403 | `{"error":"column \"x\" not allowed"}` | Column denied by policy | | 403 | `{"error":"aggregation \"x\" not allowed"}` | Aggregation fn denied by policy | | 404 | `{"error":"unknown table: x"}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 1048576 bytes"}` | Request body over the 1 MiB cap | -| 400 / 403 / 502 / 503 | `{"error":"clickhouse query: …","code":"clickhouse.…","retryable":…}` | ClickHouse failed the query: a column dropped since the schema was discovered (`400 clickhouse.rejected`), the role's `max_rows_to_read`/`max_memory_usage` cap, or a `max_execution_time` no longer than `clickhouse.query_timeout` (`400 clickhouse.limit_exceeded`), ClickHouse down (`503 clickhouse.unavailable`, `Retry-After: 5`), … — see [ClickHouse errors on the query paths](#clickhouse-errors-on-the-query-paths) | +| 400 / 403 / 502 / 503 | `{"error":"Code: 60. DB::Exception: …","code":"clickhouse.…","retryable":…}` | ClickHouse failed the query: a column dropped since the schema was discovered (`400 clickhouse.rejected`), the role's `max_rows_to_read`/`max_memory_usage` cap, or a `max_execution_time` no longer than `clickhouse.query_timeout` (`400 clickhouse.limit_exceeded`), a response over the 64 MiB read cap (`502 clickhouse.response_too_large`; earlier versions had no cap here), ClickHouse down (`503 clickhouse.unavailable`, `Retry-After: 5`), … — see [ClickHouse errors on the query paths](#clickhouse-errors-on-the-query-paths) | | 500 | `{"error":"…","code":"clickhouse.unknown","retryable":true}` | A failure with no verdict | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet, so whether the table exists is not known; `Retry-After: 5` | | 503 | `{"error":"no ClickHouse connection is open for this tenant"}` | The tenant is on no ClickHouse pool — [no pool could be opened for it](/settings-directory#clickhouse), such as one the connection ceiling refused — so the query cannot run; decided before a cached result is served, so nothing cached before is served either; `Retry-After: 30`, a settings reload retries the pool | @@ -625,7 +627,7 @@ The POST parameter body is capped at 1 MiB; a body over the cap is rejected with | 400 | `{"error":"parameter \"x\": unsupported parameter type object"}` | A non-scalar value with no SQL literal form — a JSON object, whether supplied directly or nested as an array element. A JSON **array** is valid and renders as an `IN`-style `(…)` list. | | 400 | `{"error":"parameter \"x\": array parameter must not be empty"}` | An empty array — it would render as the invalid `IN ()`. | | 413 | `{"error":"request body exceeded 1048576 bytes"}` | POST body over the 1 MiB cap | -| 400 / 403 / 500 / 502 / 503 | `{"error":"clickhouse query: …","code":"clickhouse.…","retryable":…}` | ClickHouse failed the pipe's query — for instance a parameter value it cannot use (`400 clickhouse.rejected`), or ClickHouse down (`503 clickhouse.unavailable`, `Retry-After: 5`); see [ClickHouse errors on the query paths](#clickhouse-errors-on-the-query-paths). A [pipe that writes](/pipes#pipes-that-write) answers `retryable: false` with no `Retry-After`, its message led by `clickhouse exec:` | +| 400 / 403 / 500 / 502 / 503 | `{"error":"Code: 60. DB::Exception: …","code":"clickhouse.…","retryable":…}` | ClickHouse failed the pipe's query — for instance a parameter value it cannot use (`400 clickhouse.rejected`), a response over the 64 MiB read cap (`502 clickhouse.response_too_large`; earlier versions had no cap here), or ClickHouse down (`503 clickhouse.unavailable`, `Retry-After: 5`); see [ClickHouse errors on the query paths](#clickhouse-errors-on-the-query-paths). A [pipe that writes](/pipes#pipes-that-write) answers `retryable: false` with no `Retry-After` | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | --- @@ -667,9 +669,9 @@ A raw consumer must keep the most recent announced column list and zip each `row Each SSE connection is bound to a single `?table=`; to consume multiple tables, open one connection per table. -Values of top-level `DateTime`/`DateTime64` columns inside `row` are ClickHouse's own rendering of the stored value — the exact bytes chtypes' `RowsExport` produced for that record (see [Timestamp rendering](#timestamp-rendering)), not a WaveHouse rewrite — so a live event and a `/v1/query` read of the same row agree on spelling **by construction**, with no separate canonicalization step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). A column declared with a non-UTC zone streams in that zone, not normalized to UTC; parse the timestamp with a zone-aware parser rather than assuming `Z`. +Values of top-level `DateTime`/`DateTime64` columns inside `row` are ClickHouse's own rendering of the stored value — the exact bytes chtypes' `RowsExport` produced for that record (see [Timestamp rendering](#timestamp-rendering)), not a WaveHouse rewrite — so a live event and a `/v1/query` read of the same row agree on spelling **by construction**, with no separate canonicalization step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). A column declared with a non-UTC zone is still rendered in UTC (the `Z` form), so the strings compare as the instants do. -**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause — a connection is never delivered a row the query path would hide for that role, and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: its rows are withheld with reason `unavailable` while other tenants' streams are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant's line is not served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert later fails outright at ClickHouse — a connectivity fault or batch error, not a data-shape problem chtypes would already have caught — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next live event (an in-flight gap-fill finishes under the policy snapshot taken when the stream opened), but an expired token or changed claims take effect only when the client reconnects. +**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause — a connection is never delivered a row the query path would hide for that role, and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable`, while other tenants' streams (and roles with no row filter) are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant's line is not served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert ClickHouse later rejects and parks on the dead-letter queue — a data-shape problem chtypes did not catch, since an outage only delays a row and never drops it — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), but an expired token or changed claims take effect only when the client reconnects. **CORS:** `/v1/stream` honors the request's tenant's `cors.allowed_origins` allowlist (settings directory) like every endpoint — the preflight included, which a browser sends without `X-Tenant-ID`, so over [a nested settings directory](/deployment#multi-tenant-deployments) the fronting proxy has to set the header on the `OPTIONS` too. Note that a **header-authenticated stream preflights before it connects** — `Authorization` is not CORS-safelisted — where a bare `EventSource` never preflighted at all: its request is not a `fetch()`, so Fetch's unsafe-request flag is never set and `Last-Event-ID` rides on the plain `GET`. Both headers are allow-listed, so an allowed origin connects *and* resumes cross-origin. @@ -733,7 +735,7 @@ Returns the schema for a specific table. } ``` -Per-column fields: `name`, `type` and `is_nullable` describe the column; `position` is its 1-based ordinal in the table's declaration order (always present, and the order `columns` itself is in); `has_default` says whether it declares any default at all, while `default_kind` (`DEFAULT`, `MATERIALIZED`, `ALIAS`, `EPHEMERAL`) and `default_expression` say which and what — both omitted when the column declares none. A `MATERIALIZED` or `ALIAS` column is computed and **not** insertable, so it never appears in an ingest envelope or an SSE `schema` frame, though it is still reported here and can still be selected by name. An `EPHEMERAL` column is the reverse: insertable — it exists to be written — but never stored and never selectable at all, so it can appear in a stream's column list while no query can return it. The table's `CREATE TABLE` statement is captured on the same refresh but is deliberately **not** exposed here — for a table backed by an external engine it renders that engine's wiring — endpoint, bucket/host, database, username, access key id. (ClickHouse masks the password as `[HIDDEN]` from ~23.9; the topology is what is withheld here.) +Per-column fields: `name`, `type` and `is_nullable` describe the column; `position` is its 1-based ordinal in the table's declaration order (always present, and the order `columns` itself is in); `has_default` says whether it declares any default at all, while `default_kind` (`DEFAULT`, `MATERIALIZED`, `ALIAS`, `EPHEMERAL`) and `default_expression` say which and what — both omitted when the column declares none. A `MATERIALIZED` or `ALIAS` column is computed and **not** insertable, so it never appears in an ingest envelope or an SSE `schema` frame, though it is still reported here and can still be selected by name. An `EPHEMERAL` column is the reverse: ClickHouse accepts it in an `INSERT` — it exists to be written — and WaveHouse ingest accepts it wherever the format names columns (the JSON family and `…WithNames`), feeding the `DEFAULT` expressions that read it, but it is never stored, never selectable and never published, so it appears in no envelope, no stream column list and no query result. The table's `CREATE TABLE` statement is captured on the same refresh but is deliberately **not** exposed here — for a table backed by an external engine it renders that engine's wiring — endpoint, bucket/host, database, username, access key id. (ClickHouse masks the password as `[HIDDEN]` from ~23.9; the topology is what is withheld here.) **Error responses:** @@ -894,7 +896,7 @@ The request that produced this envelope omitted `received_timestamp` (`DEFAULT n | `scope` | string | Reserved; currently always empty. | | `received_timestamp` | string | RFC 3339 nano timestamp when WaveHouse received the event. | | `format` | string | Row format. Always `JSONCompactEachRow` today; stated on the wire so a reader can tell an envelope it understands from one it doesn't. | -| `columns` | string[] | The table's **wire** column names, in declaration order — what each position in `row` means (`internal/typelayer.Table.WireColumns`: the insertable subset minus any `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column — none of the three can be named in an `INSERT`, or is ever part of a published row). | +| `columns` | string[] | The table's **wire** column names, in declaration order — what each position in `row` means (`internal/typelayer.Table.WireColumns`: the table's columns minus any `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column and minus any the role may not write — none of the three kinds is ever part of a published row). | | `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order — the exact bytes ClickHouse's own writer produced for this stored row (`internal/typelayer`'s `Table.IngestWith`, via chtypes). A column the request body omitted carries its evaluated `DEFAULT` (or the type's implicit zero value where none is declared), not `null` — the same as a native `INSERT` naming fewer columns than the table has. `DateTime`/`DateTime64` values are ClickHouse's own rendering (see [Timestamp rendering](#timestamp-rendering)), and an out-of-range integer is wrapped the way a real `INSERT` wraps it. | `columns` and `row` are only meaningful together: a reader that cannot pair them — a length mismatch, an undecodable row, a `columns` list naming one column twice — has no way to map a value to a column. Both readers also refuse an envelope whose `format` they do not recognize. Either way the SSE fan-out withholds such an envelope rather than guess, and the batch consumer parks it on the DLQ with `X-DLQ-*` headers — acking and dropping it only where the DLQ is switched off for that table, since it can never insert on retry. Both outcomes increment `wavehouse_ingest_poison_total`, separated by its `disposition` label (`parked` / `dropped`). @@ -915,7 +917,7 @@ Three values, where the envelope above has four: this is the frame a role restri ## Dead Letter Queue (DLQ) -When ClickHouse **rejects** a batch insert (a value it cannot parse, a type mismatch, a table or column it does not have), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A ClickHouse that **cannot take** the insert — down, unreachable, timing out, overloaded, read-only, or refusing WaveHouse's credentials — never sends a row here: the batch stays in the tenant's ingest queue and is retried with backoff until it inserts (see [Ingest Pipeline](/ingest-pipeline#when-clickhouse-cannot-take-an-insert)). A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: malformed JSON, an envelope of an unknown `format`, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values exactly as published: ClickHouse's own rendering of the stored value, since chtypes already validated and coerced the record before it was ever published — see [Timestamp rendering](#timestamp-rendering)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. Because chtypes catches the type and shape problems synchronously at ingest, a row that reaches this DLQ path failed for a reason chtypes couldn't have caught up front — a ClickHouse-side outage or a genuine insert-time fault — not a data mismatch. +When ClickHouse **rejects** a batch insert (a value it cannot parse, a type mismatch, a table or column it does not have), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A ClickHouse that **cannot take** the insert — down, unreachable, timing out, overloaded, read-only, or refusing WaveHouse's credentials — never sends a row here: the batch stays in the tenant's ingest queue and is retried with backoff until it inserts (see [Ingest Pipeline](/ingest-pipeline#when-clickhouse-cannot-take-an-insert)). A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: malformed JSON, an envelope of an unknown `format`, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values exactly as published: ClickHouse's own rendering of the stored value, since chtypes already validated and coerced the record before it was ever published — see [Timestamp rendering](#timestamp-rendering)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. Because chtypes catches the type and shape problems synchronously at ingest, a row that reaches this DLQ path is one ClickHouse later rejected for a reason chtypes couldn't have caught up front, not a data mismatch; an outage only delays a row and never parks it. Under [`mq.backend: nats`](/deployment#external-nats) the parked rows of every tenant go to one shared dead-letter stream instead, under `.dlq.{tenant}.{table}`; the bodies and headers are the same. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index a99acda0..a7c7a67d 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -45,7 +45,7 @@ flowchart TD ## Binaries -WaveHouse ships a single binary, `wavehouse`: an all-in-one process running the API, batch worker, embedded NATS JetStream, and optional embedded Pebble dedup. The binary also loads a second artifact at start: a per-ClickHouse-version shared library (`internal/typelayer`, via [chtypes](/deployment#chtypes-artifacts)) that runs ClickHouse's own parser in-process for ingest validation, type coercion, and row-level security, and only a process running the `api` role loads it. The binary requires cgo (dlopen only — no static link to the artifact) and glibc, so supported platforms are Linux amd64/arm64 and macOS arm64. The only external network dependency is ClickHouse, unless a shared backend is selected: `cache.backend: redis`, `dedupe.backend: dynamodb`, or `mq.backend: nats`, which points the queue at a NATS cluster the operator runs and lets several processes, each running some of the [roles](/configuration#process-roles), share it. `cmd/wavehouse` is the shell — subcommand dispatch, the logger, `config.Load`, the signal context — and `internal/app` is the process itself (see [`app/`](#app--process-wiring) below). +WaveHouse ships a single binary, `wavehouse`: an all-in-one process running the API, batch worker, embedded NATS JetStream, and optional embedded Pebble dedup. The binary also loads a second artifact at start: a per-ClickHouse-version shared library (`internal/typelayer`, via [chtypes](/deployment#chtypes-artifacts)) that runs ClickHouse's own parser in-process for ingest validation, type coercion, and row-level security, and only a process running the `api` role loads it. The binary requires cgo (dlopen only — no static link to the artifact) and, on Linux, glibc 2.34 or later, so supported platforms are Linux amd64/arm64 and macOS arm64. The only external network dependency is ClickHouse, unless a shared backend is selected: `cache.backend: redis`, `dedupe.backend: dynamodb`, or `mq.backend: nats`, which points the queue at a NATS cluster the operator runs and lets several processes, each running some of the [roles](/configuration#process-roles), share it. `cmd/wavehouse` is the shell — subcommand dispatch, the logger, `config.Load`, the signal context — and `internal/app` is the process itself (see [`app/`](#app--process-wiring) below). ## Internal Packages @@ -85,8 +85,8 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. - **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch, and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read and ahead of every other tenant's traffic. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). -- **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` — a deliberately different audience from the structured-query path, which leaves ClickHouse's default spelling alone so it matches the SSE wire. -- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=simple`) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. Connections per tenant are capped at the tenant's `max_open_conns`. +- **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso`, the same spelling the structured-query path and the SSE wire use, so a timestamp reads the same on every surface. +- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=iso`, so a timestamp is RFC 3339 in UTC and matches the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. Each tenant's connections are capped at its `max_open_conns`, per pool identity (URL, user, database and TLS settings): two tenants that share all four share one reader and one cap, and tenants that differ in any of them never share. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) chconn.Target` beside it — each resolves the request's tenant per call, and the zero target (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for one tenant's per-table parked counts (optionally one table) and its total — the tenant `?tenant=` names, read strictly by `opsTenant`, tenant `0` without it. The tenant is looked up in the MQ, not the settings registry, so a rejected or removed tenant's parked rows are read like a served one's; a tenant with no dead-letter queue (`mq.ErrNoDeadLetterQueue`) is a 404, and any other failure to read it a 500. The queue itself is `internal/mq`'s. @@ -158,12 +158,12 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `discovery/` — Schema Discovery -- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant to another user or `tls` block, which reads the same tables, applies to the next refresh, one that moves it to another address or database has `internal/app`'s `discoveries` drop its registry and build a fresh one, whose first discovery runs at once and whose lookups answer `ErrNotLoaded` until it succeeds ([#638](https://github.com/Wave-RF/WaveHouse/issues/638)), and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) can decide the insertable subset the ingest envelope and the SSE announcement are both built from. Each refresh also records the server version (`SELECT version()`) and default time zone (`SELECT timezone()`, exposed via `ServerTimezone()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), and reads each column's `default_expression` and 1-based `position` alongside its type. Overlapping refreshes (the loop and an on-demand one) publish in the order they started: a refresh that finishes after a later-started one has published returns without publishing, so an older snapshot never replaces a newer one. An `OnRefresh` hook fires synchronously after every refresh that publishes and **before** the registry reports itself loaded, so a loaded tenant is a bound one — `internal/typelayer.Engine.Bind` is its only registered consumer, and it is what resolves the chtypes artifact matching that tenant's server line and recompiles its per-table handles ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop, each sleep drawn uniformly below the backoff so instances retrying one ClickHouse do not retry in lockstep, used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. +- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant to another user or `tls` block, which reads the same tables, applies to the next refresh, one that moves it to another address or database has `internal/app`'s `discoveries` drop its registry and build a fresh one, whose first discovery runs at once and whose lookups answer `ErrNotLoaded` until it succeeds ([#638](https://github.com/Wave-RF/WaveHouse/issues/638)), and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) describe the table's writable columns; the ingest envelope and the SSE announcement are built from the type layer's `Table.WireColumns` instead, which also leaves out `EPHEMERAL` columns. Each refresh also records the server version (`SELECT version()`) and default time zone (`SELECT timezone()`, exposed via `ServerTimezone()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), and reads each column's `default_expression` and 1-based `position` alongside its type. Overlapping refreshes (the loop and an on-demand one) publish in the order they started: a refresh that finishes after a later-started one has published returns without publishing, so an older snapshot never replaces a newer one. An `OnRefresh` hook fires synchronously after every refresh that publishes and **before** the registry reports itself loaded, so a loaded tenant is a bound one — `internal/typelayer.Engine.Bind` is its only registered consumer, and it is what resolves the chtypes artifact matching that tenant's server line and recompiles its per-table handles ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop, each sleep drawn uniformly below the backoff so instances retrying one ClickHouse do not retry in lockstep, used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. - **discovery_test.go** — Unit tests for schema discovery. ### `typelayer/` — In-Process ClickHouse Parser -The only package that imports `github.com/wave-rf/chtypes/go/chtypes` (the sole exception: `cmd/wavehouse/main.go` references `typelayer` itself, never chtypes directly). One process-wide `Engine` wraps one lazily-opened `chtypes.Registry`, found through a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`; empty means the chtypes search path), and only a process running the `api` role builds it — an ingest-worker-only or sweeper-only process never loads the artifact and boots without one. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for what ships where and how large it is. +The only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` wraps one lazily-opened `chtypes.Registry`, found through a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`; empty means the chtypes search path), and only a process running the `api` role builds it — an ingest-worker-only or sweeper-only process never loads the artifact and boots without one. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for what ships where and how large it is. Each tenant has its own table set inside that engine, bound from the tenant's own schema refresh and released when the tenant is no longer served. @@ -171,7 +171,8 @@ Each tenant has its own table set inside that engine, bound from the tenant's ow - **`Engine.RoleTable(tenant, table, shape)`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column is simply not in the schema, so naming it is ClickHouse's code 117, and an absent check column takes the claim as its default while a supplied value still wins. - **`Table.IngestWith(format, opts, body, checks...)`** (`Ingest` is the same with no options) runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. - **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per table. Only a definite true admits; a predicate error, a policy column the table no longer has, schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. -- A tenant that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row for that tenant's tables with reason `unavailable`. +- A tenant that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row of that tenant's tables from a role that has a row `filter`, with reason `unavailable`. +- A role whose column policy cannot be compiled is refused with a non-retryable error and the cause is logged; it is not the `503` an unavailable tenant gets, since retrying cannot fix a policy. See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest error-response shape and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced) for how predicates are compiled and evaluated. @@ -282,8 +283,9 @@ Client POST /v1/ingest?table={table} request: ClickHouse's own parser type-checks, coerces, and fills DEFAULTs (including now()) per record. A rejected record carries ClickHouse's real error code (exception_code) and message — an unknown field, a - MATERIALIZED/ALIAS/EPHEMERAL column, or a column this role may not write is - 117; a record chtypes cannot answer for is a distinct "declined" outcome + MATERIALIZED/ALIAS column, or a column this role may not write is + 117 (an EPHEMERAL value is accepted where the format names columns and + feeds DEFAULTs, never stored or published); a record chtypes cannot answer for is a distinct "declined" outcome (422), never a data rejection → Evaluate the role's check clauses over the accepted rows with one compiled chtypes filter — false is 403 for that record, unevaluable is 422 diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index f410be44..53429b57 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -162,22 +162,22 @@ The server speaks **plain HTTP** — there is no inbound-TLS setting. Terminate ### ClickHouse -Only the secret and the connection ceiling are boot config. The wiring — native address, HTTP port and scheme, database, username, query timeout, TLS, headers, pool sizes — is the settings directory's [`clickhouse` block](/settings-directory#clickhouse), where a change re-dials without a restart. +Only the secret, the connection ceiling and the chtypes artifact directory are boot config. The wiring — native address, HTTP port and scheme, database, username, query timeout, TLS, headers, pool sizes — is the settings directory's [`clickhouse` block](/settings-directory#clickhouse), where a change re-dials without a restart. | YAML Key | Env Var | Default | Description | | --- | --- | ------- | ----------- | | `clickhouse.password` | `WH_CH_PASSWORD` | *(empty)* | Authentication password, combined with the settings directory's `clickhouse.username` on every (re)connect. A secret, so it never lives in a tracked JSON file; rotating it is a restart. | -| `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the `max_open_conns` of the open pools — one per distinct connection tuple among the served tenants, see [the settings directory](/settings-directory#clickhouse) — must not add up to more. Pools above it at boot refuse to start, naming the sum and the ceiling; on a reload a pool resized above it is refused and keeps its size, and a new pool that would cross it is not opened — its tenants keep the pool they had, or have none — both logged (the reload itself still reports `adopted`) and retried by the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. | +| `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the `max_open_conns` of the open pools — one per distinct connection tuple among the served tenants, see [the settings directory](/settings-directory#clickhouse) — must not add up to more. Pools above it at boot refuse to start, naming the sum and the ceiling; on a reload a pool resized above it is refused and keeps its size, and a new pool that would cross it is not opened — its tenants keep the pool they had, or have none — both logged (the reload itself still reports `adopted`) and retried by the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. It counts native connections only: the HTTP reader behind structured queries and pipes is capped separately, per pool identity (URL, user, database and TLS settings), by that pool's `max_open_conns`. | | `clickhouse.chtypes_registry` | `WH_CHTYPES_REGISTRY` | *(empty)* | Directory holding the chtypes artifacts (one `/` per ClickHouse line). Empty defers to the chtypes search path — `$CHTYPES_REGISTRY`, `~/.cache/chtypes/artifacts/abi6/-`, then the system directories. Either way a library is opened lazily, on first use of its line; an explicit directory is searched first, then the rest of the path. Boot-tier: changing it is a restart. See [chtypes artifacts](/deployment#chtypes-artifacts). | ### ClickHouse user access The ClickHouse user WaveHouse connects as (the settings directory's `clickhouse.username`) needs these, and nothing about the tables themselves has to be exposed to anyone else: -- **`SELECT` on the tenant's tables.** Structured queries, read pipes and the stream's replay read through it. +- **`SELECT` on the tenant's tables.** Structured queries and read pipes read through it. - **`INSERT` on the tenant's tables, for the ingest worker.** The worker writes each batch with `INSERT … FORMAT JSONCompactEachRow` over the HTTP interface. A process that only serves reads never inserts, but the same user is used by every process, so grant it wherever any process ingests. - **Read access to `system.columns` and `system.tables`**, plus `SELECT timezone()` and `SELECT version()`, which need no grant. Schema discovery reads the columns, their `DEFAULT`/`MATERIALIZED` kinds and each table's DDL, and records the server's version and time zone; the version picks the [chtypes artifact](/deployment#chtypes-artifacts) and the zone is the one the in-process parser reads bare timestamps in. ClickHouse lists only the tables a user holds a grant on, so a table the user cannot see is a table WaveHouse does not know. -- **A profile that lets it change per-query settings.** WaveHouse sends settings with every query: reads carry `readonly=2` together with `max_execution_time`, `wait_end_of_query`, `http_write_exception_in_output_format`, `cancel_http_readonly_queries_on_client_close` and the JSON/date rendering settings, and the ingest worker's `INSERT` carries `date_time_input_format`, `input_format_null_as_default` and `async_insert=0`. A profile with `readonly=1` (or a `` entry that pins any of these) makes ClickHouse refuse them: a read answers `502 clickhouse.misconfigured` (the `READONLY` refusal), and the worker's batches fail. Use `readonly=0`, or `readonly=2` for a user that only ever reads: `readonly=2` still forbids the ingest worker's `INSERT`. +- **A profile that lets it change per-query settings.** WaveHouse sends settings with every query: reads carry `readonly=2` together with `max_execution_time`, `wait_end_of_query`, `http_write_exception_in_output_format`, `cancel_http_readonly_queries_on_client_close` and the JSON/date rendering settings, and the ingest worker's `INSERT` carries `date_time_input_format`, `input_format_null_as_default` and `async_insert=0`. A profile with `readonly=1` (or a `` entry that pins any of these) makes ClickHouse refuse them: a read refused with `READONLY` answers `502 clickhouse.misconfigured`, a `` entry that a setting violates is refused as `400 clickhouse.rejected`, and the worker's batches fail. Use `readonly=0`, or `readonly=2` for a user that only ever reads: `readonly=2` still forbids the ingest worker's `INSERT`. - **Whatever else the statements you hand it need.** A [pipe that writes](/pipes#pipes-that-write) and the admin-only raw SQL endpoint `/v1/ops/query` run as this user too, so a pipe that runs `ALTER` needs the grant for it, and a missing grant is `403 clickhouse.access_denied`. ### Server-side resource limits @@ -348,6 +348,7 @@ clickhouse: password: "" # addr, ports, database, username, query_timeout, tls, # headers and pool sizes are settings (config.json) max_total_conns: 0 # ceiling on open native connections; 0 = none + chtypes_registry: "" # chtypes artifact directory; empty = the SDK's search path mq: backend: embedded # in-process NATS JetStream under /nats @@ -449,6 +450,8 @@ WH_SERVER_SHUTDOWN_TIMEOUT=10 WH_CH_PASSWORD= WH_CH_MAX_TOTAL_CONNS=0 +# chtypes artifact directory; empty = the SDK's search path +WH_CHTYPES_REGISTRY= WH_MQ_BACKEND=embedded # With WH_MQ_BACKEND=nats (read only then): diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 2104e0d5..baa2ccf5 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -76,13 +76,13 @@ WH_SERVER_PORT=9090 \ docker build -f deployments/Dockerfile -t wavehouse:latest . ``` -This builds the runtime image `wavehouse:latest`. (The published `ghcr.io` images are built by GoReleaser from `deployments/Dockerfile.goreleaser`, not this command — see Registry below.) +This builds the runtime image `wavehouse:latest`. (The published `ghcr.io` images are built by the release workflows with `docker buildx` from `deployments/Dockerfile.goreleaser`, not this command — see Registry below.) All images use multi-stage builds (`golang:1.27-bookworm` glibc builder → `gcr.io/distroless/cc-debian12` runtime — cgo needs glibc, so the previous Alpine/musl builder and `distroless/static` runtime no longer work) for minimal attack surface. The build also fetches the [chtypes artifact(s)](#chtypes-artifacts) pinned in `chtypes.lock` into the image. ### Registry -Production images are published to GitHub Container Registry via GoReleaser: +Production images are published to GitHub Container Registry by the release workflows: ```text ghcr.io/wave-rf/wavehouse: @@ -112,18 +112,18 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ **Which processes load it.** Only processes with the `api` role — the ones that serve ingest and the stream. A process that runs only the ingest worker or the sweeper loads no artifact and boots without one installed. An API process refuses to start when no artifact is installed at all. -**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and the stream withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with the cause in the server log; run such tenants in a separate process, or align the servers' `timezone` setting. See [API → Ingest error responses](/api#error-responses) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). +**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with the cause in the server log; run such tenants in a separate process, or align the servers' `timezone` setting. See [API → Ingest error responses](/api#error-responses) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). **Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse does not autofetch on a miss in production — an unmatched line is a boot-time or refresh-time failure, not a background download. **Size.** Each artifact is roughly 160–290 MB on disk; a running process holding several loaded versions (e.g. across a rolling ClickHouse upgrade) costs roughly 120 MB of resident memory per loaded version (the chtypes multi-version guide's figure; a library is opened on first use of its line, not at registry construction). -**Docker images** ship the artifact(s) baked in: the image build fetches whatever `chtypes.lock` names (see below), so a container never needs network access to ClickHouse's artifact store at runtime. `WH_CHTYPES_REGISTRY` (default `/opt/chtypes/artifacts`) points at the directory inside the image. +**Docker images** ship the artifact(s) baked in: the image build fetches whatever `chtypes.lock` names (see below), so a container never needs network access to chtypes' artifact store at runtime. The image sets `CHTYPES_REGISTRY=/opt/chtypes/artifacts` (the SDK's own variable); set `WH_CHTYPES_REGISTRY` only to point at a bind-mounted directory instead. **The published images support ClickHouse 26.8 only**: they bake the lines `chtypes.lock` names, and a server on any other line answers every tenant on it `503` (with row-filtered streams withholding their rows) behind a generic body. For another line, fetch its artifact (below), mount the directory into the container and set `WH_CHTYPES_REGISTRY` to the mount path. -**Release archives and `go install` / building from source** do not carry or fetch an artifact — only the Docker images bake one in. See the [README's `go install` caveat](https://github.com/Wave-RF/WaveHouse#c-go-install-binary-no-docker). Fetch one yourself before first run: +**Release archives and `go install` / building from source** do not carry or fetch an artifact — only the Docker images bake one in. See the [README's `go install` caveat](https://github.com/Wave-RF/WaveHouse#c-go-install-binary-no-docker). Fetch one yourself before first run. Both routes below run the chtypes command with `go run`, which needs Go 1.27 and a C compiler; a release archive has neither `scripts/fetch-chtypes.sh` nor the lock file, so use the second form there, or fetch from a checkout and copy the directory to the host: ```bash -scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --frozen --lock chtypes.lock 26.8 +scripts/fetch-chtypes.sh # from a checkout; wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --frozen --lock chtypes.lock 26.8 ``` or, for a line not in the repo's lock file: @@ -170,12 +170,14 @@ All configuration can be set via environment variables. This is the recommended Key variables for production: ```bash -# ClickHouse: only the password and the connection ceiling are env. The +# ClickHouse: only the password, the connection ceiling and the chtypes +# artifact directory are env. The # address, HTTP port/scheme, database, user, TLS, headers and pool sizes are # clickhouse.* in the settings directory's config.json. WH_CH_PASSWORD= # Ceiling on open native ClickHouse connections; 0 = none # WH_CH_MAX_TOTAL_CONNS=0 +# WH_CHTYPES_REGISTRY= # empty unless you bind-mount the artifacts (see chtypes artifacts) # Auth secrets (the JWT middleware always runs — set a secret, or auth.jwks_url # in the settings directory, to validate tokens; without one, every request @@ -643,7 +645,7 @@ For local development, `docker compose -f deployments/compose/dependencies.yaml WaveHouse uses a **Bring Your Own Schema** model. You create your tables in ClickHouse with whatever columns and engines you need. WaveHouse discovers the schemas automatically via `system.columns` and validates ingest data against them — see [Schema Validation](/api#post-v1ingesttabletable--ingest-data) for the rules a record must satisfy. -Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names one is rejected with ClickHouse's own code (117) rather than published; a policy `check` naming one is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's implicit zero value where none is declared, evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). +Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted where the format names columns (the JSON family, `…WithNames`) and feeds the `DEFAULT`s that read it without being stored or published; a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's implicit zero value where none is declared, evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). Example table: @@ -754,9 +756,8 @@ The streaming surface loses something too, more quietly. SSE gap-fill (`?since=` **The upgrade does not carry the old queue over at all.** Boot deletes the earlier build's queue and dead-letter queue (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) and everything in them, logging a `WARN` with each one's message count: an event the old build had not yet inserted, and a row it had already parked, do not survive the upgrade. Draining first keeps the events not yet inserted; a row already parked is lost with the queue, since the earlier build offers no way to read one back (`GET /v1/ops/dlq/stats` returns counts only). -Three audits belong **before** the drain, because none of them announces itself afterwards: +Two audits belong **before** the drain, because none of them announces itself afterwards: -- **The wire envelope's `row` is positional** (`columns` names each slot), so a message from before this migration and one from after it look the same shape-wise; what changed underneath is how an omitted field is resolved into that slot — see [the ingest note](/ingest-pipeline#the-journey-of-one-event) for the current behavior. - **Policy `check` blocks are now validated against the table.** A `check` naming a column the table lacks, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is a per-record `403` on *every* insert by that role. `wavehouse validate` cannot catch it — it never sees the ClickHouse schema — so audit them against their tables first. See [Access control → Insert checks](/access-control#insert-checks). - **Every `WH_*` variable the binary does not bind refuses boot.** The old binary ignored a variable it did not read; the new one names every unbound one and exits before it opens the queue, so a pod spec or compose file that still carries one comes back from the upgrade as a container that will not start. Diff the environment against the [Configuration Reference](/configuration) first: a `WH_*` variable that is not in its tables is unbound, and whatever it used to configure now lives in the [settings directory](/settings-directory) or is gone. A Kubernetes Service in the pod's namespace named `wh` or `wh-*` counts too: it injects link variables under the `WH_` prefix (`WH_SERVICE_HOST` and `WH_PORT` for `wh`, `WH_FOO_SERVICE_HOST` and `WH_FOO_PORT` for `wh-foo`), so set `enableServiceLinks: false` on the pod spec. diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index 7aa7af9e..b26ec107 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -13,7 +13,7 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: | Tool | Required version | Why | Install | | ---- | ---------------- | --- | ------- | -| **Go** | 1.27+ (matches `go.mod`) | Compiles `cmd/wavehouse` with cgo enabled (needed by chtypes' dlopen shim — a C toolchain and glibc must be present); also runs the pinned `tool` deps (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `deadcode`, `gsa`, `goda`) via `go tool` | [go.dev/dl](https://go.dev/dl/) | +| **Go** | 1.27+ (matches `go.mod`) | Compiles `cmd/wavehouse` with cgo enabled (needed by chtypes' dlopen shim — a C toolchain, and on Linux glibc 2.34 or later, must be present); also runs the pinned `tool` deps (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `deadcode`, `gsa`, `goda`) via `go tool` | [go.dev/dl](https://go.dev/dl/) | | **GNU Make** | **4.0+** | The Makefile uses `--output-sync=target` (Make 4 only) and bash-pinned recipes. macOS ships with BSD Make 3.81, which **will not work** | macOS: `brew install make` then use `gmake` or put `$(brew --prefix make)/libexec/gnubin` on your PATH. Linux: usually already installed | | **bash** | 4+ recommended | Recipes are pinned to `bash`; the helper scripts under `scripts/` use `set -euo pipefail` and bash arrays | macOS default is bash 3.2 (works for current recipes, but `brew install bash` is safer); Linux distros ship 4+ | | **Docker** *(or Podman)* | Engine 20.10+ with the Compose **v2** plugin (`docker compose`, no hyphen) | Compose stacks under `deployments/compose/`; the E2E and integration suites boot ClickHouse and a Redis via testcontainers (no compose file), the integration suite also dynamodb-local, and the integration suite also runs the shared cache backend against Redis, Valkey, Dragonfly (pulled from `docker.dragonflydb.io`) and a one-node Redis Cluster | [Docker Desktop](https://docs.docker.com/get-docker/), [colima](https://github.com/abiosoft/colima), or [Podman](https://podman.io) with `podman-compose` / the `podman compose` plugin. The testcontainers Go library also honors `DOCKER_HOST` for rootless Podman setups | @@ -29,13 +29,13 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --frozen --lock chtypes.lock 26.8 ``` -It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–290 MB — expect the first run to take a minute or two. Without it, `make dev` / `make test` / `make test-e2e` fail closed (a `503` on ingest, every stream row withheld) until a matching artifact exists for the ClickHouse line the tests or your local server run against. +It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–290 MB — expect the first run to take a minute or two. Without it the API process refuses to boot (`make dev`, `make test-e2e`), and the unit tests that need the engine skip; set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (CI does) to make a missing artifact fail those tests instead. A `503` on ingest, with row-filtered stream rows withheld, is what a process that did find an artifact answers for a ClickHouse line the artifact does not cover. ### Auto-installed by `make tools` Run `make tools` once after cloning to populate everything that doesn't have to be on your PATH: -- **`golangci-lint` v2.11.4** → installed to `.bin/_/` (version-pinned in the Makefile; bumping the version triggers a reinstall). Not in `go.mod` because its dependency tree conflicts with the main module. +- **`golangci-lint` v2.13.2** → installed to `.bin/_/` (version-pinned in the Makefile; bumping the version triggers a reinstall). Not in `go.mod` because its dependency tree conflicts with the main module. - **`misspell` v0.8.0, `shellcheck` v0.11.0, `actionlint` v1.7.12** → installed to `.bin/_/`; they back `make lint-prose`, `make lint-sh`, and `make lint-gha`. `make tools` also points `core.hooksPath` at `.githooks/`, which is what installs the pre-commit and pre-push gates. - **`air` v1.65.1** → installed to `.bin/_/` via `go install`; used by `make dev` for hot-reload. Same exclusion principle as `golangci-lint` — air's transitive deps (Hugo, Sass libs) would bloat `go.sum`. - **Go `tool` deps** (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `go-test-coverage`, `gocover-cobertura`, `deadcode`, `gsa`, `goda`) — pinned in `go.mod` via native `tool` directives (Go 1.24+), invoked with `go tool `. `make tools` runs `go mod download` so they're cached; they compile lazily on first invocation. diff --git a/docs/src/content/docs/getting-started.md b/docs/src/content/docs/getting-started.md index d2b91168..24c62f55 100644 --- a/docs/src/content/docs/getting-started.md +++ b/docs/src/content/docs/getting-started.md @@ -11,7 +11,7 @@ Run WaveHouse locally in under five minutes. WaveHouse ships as one binary plus - **Docker** — for running ClickHouse (and optionally WaveHouse itself). - **curl** and **jq** (optional) — for poking the API. -- **Go 1.27+** — only required if you want to build from source; skip it for the Docker path below. Building from source also requires cgo (a C toolchain) and glibc — see [Deployment → Supported Platforms](/deployment#supported-platforms). +- **Go 1.27+** — only required if you want to build from source; skip it for the Docker path below. Building from source also requires cgo (a C toolchain; on Linux, glibc 2.34 or later) — see [Deployment → Supported Platforms](/deployment#supported-platforms). ## 1. Start WaveHouse diff --git a/docs/src/content/docs/index.mdx b/docs/src/content/docs/index.mdx index 1033bfaa..968324e1 100644 --- a/docs/src/content/docs/index.mdx +++ b/docs/src/content/docs/index.mdx @@ -60,7 +60,7 @@ import { cloudLink } from '../../config/outbound'; ## Why WaveHouse exists -ClickHouse is a phenomenal OLAP database, but pointing a frontend straight at it comes with sharp edges: custom APIs, Kafka queues to avoid "too many parts" errors, replacing-merge logic for deduplication. WaveHouse abstracts all of that into a single, deployable binary so you stop interacting with ClickHouse directly. +ClickHouse is a phenomenal OLAP database, but pointing a frontend straight at it comes with sharp edges: custom APIs, Kafka queues to avoid "too many parts" errors, replacing-merge logic for deduplication. WaveHouse abstracts all of that into a single, deployable process so you stop interacting with ClickHouse directly.
@@ -88,7 +88,7 @@ If you're building user-facing analytics, **WaveHouse is like Supabase for Click - WaveHouse discovers your ClickHouse schemas via `system.columns` and validates every ingest against the real schema — unknown fields, type mismatches, and null violations are rejected at the edge. + WaveHouse discovers your ClickHouse schemas via `system.columns` and validates every ingest against the real schema — unknown fields and values ClickHouse cannot parse are rejected at the edge. Writes land in a durable NATS JetStream WAL and return `200 OK` instantly. A background worker batch-flushes to ClickHouse — never drop a packet. @@ -180,11 +180,11 @@ curl -N "http://localhost:8080/v1/stream?table=clicks" WaveHouse is fail-closed, so the standalone stack ships a permissive trial policy (a non-admin `public` role that can read/write the demo tables, in the bind-mounted settings directory) — then it's five minutes from `git clone` to a live event stream. See [Getting Started](/getting-started). -**How it ships:** one `wavehouse` process — API, batch worker, embedded NATS JetStream, and optional embedded Pebble dedup, all in-process. **The only external dependency is ClickHouse.** Same binary on a laptop, in Docker, or on a production host. +**How it ships:** one `wavehouse` process — API, batch worker, embedded NATS JetStream, and optional embedded Pebble dedup, all in-process. **The only external service is ClickHouse**; the process also loads one local file, the [chtypes artifact](/deployment#chtypes-artifacts) for your ClickHouse line, which the Docker images carry. Same binary on a laptop, in Docker, or on a production host. ## Or skip the ops entirely -Self-hosting WaveHouse is deliberately boring — one binary, one dependency. But *that one dependency* is a ClickHouse cluster somebody has to size, upgrade, back up, and keep alive at 3am. +Self-hosting WaveHouse is deliberately boring — one process, one service dependency. But *that one dependency* is a ClickHouse cluster somebody has to size, upgrade, back up, and keep alive at 3am. ` prefix is looked through: the statement after it is the one classified. Such a pipe runs on **every** call: it never reads or fills the cache and is never coalesced with an identical call in flight, so ten identical calls are ten writes. The response is `[]` with `X-Cache: BYPASS` and `Cache-Control: no-store`, so an HTTP cache in front of a `GET` does not answer a repeat either. WaveHouse classifies the statement by its leading keyword — a `WITH`-led one by whether it holds `INSERT INTO` outside parentheses — with the same classifier that sends it to ClickHouse as a write, so no pipe property marks it. A statement led by any other keyword runs as a read: one that returns rows (`BACKUP`, `RESTORE`) is cached and coalesced, so a repeat within the TTL does not run; one that returns none (`UNDROP`, `MOVE`) fails the call after it has run, which the SDK may retry — don't put a write led by another verb in a pipe ([#666](https://github.com/Wave-RF/WaveHouse/issues/666)). +A pipe's SQL may be a write: a statement led by a write verb WaveHouse recognizes — `INSERT`, `UPDATE`, `DELETE`, `ALTER` (so `ALTER … DELETE`), `CREATE`, `DROP`, `TRUNCATE`, `RENAME`, `EXCHANGE`, `REPLACE`, `OPTIMIZE`, `ATTACH`, `DETACH`, `GRANT`, `REVOKE`, `KILL`, `SET`, `USE` or `SYSTEM` — directly, or `INSERT INTO` after a `WITH` list (`WITH … INSERT INTO …`, the only write ClickHouse accepts there). An `EXECUTE AS ` prefix is looked through: the statement after it is the one classified. Such a pipe runs on **every** call: it never reads or fills the cache and is never coalesced with an identical call in flight, so ten identical calls are ten writes. The response is `[]` with `X-Cache: BYPASS` and `Cache-Control: no-store`, so an HTTP cache in front of a `GET` does not answer a repeat either. WaveHouse classifies the statement by its leading keyword — a `WITH`-led one by whether it holds `INSERT INTO` outside parentheses — with the same classifier that sends it to ClickHouse as a write, so no pipe property marks it. A statement led by any other keyword is sent as a read, and a read runs under `readonly=2`: ClickHouse refuses a data-changing statement such as `BACKUP`, `RESTORE`, `UNDROP` or `MOVE` before running it, and the call answers `502 clickhouse.misconfigured`, which is not retryable. Nothing runs and nothing is cached, but the pipe never works — don't put a write led by another verb in a pipe ([#666](https://github.com/Wave-RF/WaveHouse/issues/666)). A failed write is not retried automatically, because it may have run. It answers with the status and `code` a failed read would ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)), but always with `retryable: false` and no `Retry-After`, `503 clickhouse.unavailable` included: once the statement is on its way to ClickHouse, WaveHouse cannot tell whether it ran. The [SDK](/sdk/pipes) does not retry such an answer, so check whether the write landed before you send it again. A call refused before anything is sent — the tenant on no ClickHouse pool, `503` with `Retry-After: 30` — cannot have run, and the SDK retries it. The SDK also retries when WaveHouse's own answer never reaches it — a dropped connection, or a `502`/`503`/`504` from a proxy in front of WaveHouse that gave up waiting — so a write can still run twice that way; give a client that runs write pipes [`options.maxRetries`](/sdk#clientconfigdb) `0` if that matters. diff --git a/docs/src/content/docs/reverse-proxy.mdx b/docs/src/content/docs/reverse-proxy.mdx index 5651d531..04331df8 100644 --- a/docs/src/content/docs/reverse-proxy.mdx +++ b/docs/src/content/docs/reverse-proxy.mdx @@ -93,7 +93,7 @@ WaveHouse caps the inbound request body it will decode, as a memory-safety backs | `POST /v1/query`, `GET/POST /v1/pipes/{name}` (parameter / AST bodies) | **1 MiB** | `413 {"error":"request body exceeded 1048576 bytes"}` | | `POST /v1/ingest`, `POST /v1/ops/query` (bulk payload bodies) | **16 MiB** | `413 {"error":"request body exceeded 16777216 bytes"}` | -These caps are **fixed and not configurable** — they aren't a tuning knob, they're an invariant. A JSON request body amplifies roughly an order of magnitude when decoded into memory (a large array or object of small values explodes into Go's in-memory representation), so an *uncapped* decoder on a public endpoint is a single-request out-of-memory vector. A query or pipe-parameter body is bounded by nature — a real one is far under 1 MiB even with a large `in`-list — so the 1 MiB cap is generous headroom that never binds legitimate use. +These caps are **fixed and not configurable** — they aren't a tuning knob, they're an invariant. A JSON request body amplifies roughly an order of magnitude when decoded into memory (a large array or object of small values explodes into Go's in-memory representation), so an *uncapped* decoder on a public endpoint is a single-request out-of-memory vector. A query or pipe-parameter body is bounded by nature — a real one is far under 1 MiB even with a large `in`-list — so the 1 MiB cap is generous headroom for ordinary use. A structured query with a very large `in` list is the case that can reach it, since [the list travels in the body](/api#post-v1querytabletable--structured-query), so size such lists with the cap in mind. Set your own **outer** limit at the proxy, sized to your real needs: diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index bc9b319e..96b6bcdb 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -37,8 +37,8 @@ The SDK **never throws** for anything the server returns — all API errors come | 403 | `clickhouse.access_denied` | No | ClickHouse's user lacks a grant the statement needs | | 500 | `HTTP_500` | Yes | Server error (retried per `maxRetries`) | | 500 / 502 | `clickhouse.unknown` | Yes | ClickHouse failed with no verdict (no exception code, no recognizable transport error); `502` on `wh.sql` | -| 502 | `clickhouse.misconfigured` | No | ClickHouse refused WaveHouse's own credentials or database, or the route to it is wrong (a redirect, or a `4xx` other than `408`/`413`/`429`, with no exception code) — an operator fix | -| 502 | `clickhouse.response_too_large` | No | A raw-SQL (`wh.sql`) response over the 64 MiB cap | +| 502 | `clickhouse.misconfigured` | No | ClickHouse refused WaveHouse's own credentials or database, a read ran under a `readonly=1` profile and was refused as a write, or the route to it is wrong (a redirect, or a `4xx` other than `408`/`413`/`429`, with no exception code) — an operator fix | +| 502 | `clickhouse.response_too_large` | No | A response over the 64 MiB cap, on any query path (structured query, pipe or `wh.sql`) | | 503 | `clickhouse.unavailable` | Yes | ClickHouse is down, unreachable or overloaded; `Retry-After: 5`, honored between attempts | | 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, a schema not discovered yet, a tenant on no ClickHouse pool, a tenant whose ClickHouse line the type layer cannot serve (`ingest validation is unavailable`, `Retry-After: 5`), a dedupe store that cannot answer (`dedupe store unavailable`, `Retry-After: 5`), a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`), or a record whose dedupe id another request is still publishing (`a request with the same dedupe id is in flight`, `Retry-After`: the server's dedupe lease, 30 s by default). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on those last two causes waits that long; a stream re-dials on its own jittered backoff instead | | 0 | `NETWORK_ERROR` | Yes | Network failure (retried with exponential backoff) | @@ -147,7 +147,7 @@ Codegen reads `/v1/ops/schema`, which is **admin-only**. Against a non-dev serve | `--out`, `-o` | Output .d.ts file path | `./wavehouse.d.ts` | | `--auth`, `-a` | Bearer token (if auth required) | — | -The generated row type is the **read** shape, and computed columns are where it and the server disagree. An `EPHEMERAL` column declares a default, so codegen emits it, yet no query can ever return it — the type says readable where only the write is real. `MATERIALIZED` and `ALIAS` columns declare defaults too, so they are emitted as optional, but supplying any of the three on `insert` is a `400` carrying ClickHouse's own code 117 (`Unknown field found while parsing JSONEachRow format: x`), and the type will not catch it. Omit computed columns; the server fills them in. +The generated row type is the **read** shape, and computed columns are where it and the server disagree. An `EPHEMERAL` column declares a default, so codegen emits it, yet no query can ever return it — the type says readable where only the write is real. `MATERIALIZED` and `ALIAS` columns declare defaults too, so they are emitted as optional, but supplying either on `insert` is a `400` carrying ClickHouse's own code 117 (`Unknown field found while parsing JSONEachRow format: x`), and the type will not catch it. An `EPHEMERAL` value is accepted on a JSON `insert` and feeds the defaults that read it, but is never stored or returned. Omit computed columns; the server fills them in. **Example output:** diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index 92e91e12..2c803df5 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -99,7 +99,7 @@ interface StreamEvent { `data` is a row **object**, as it always has been — but the wire underneath is positional. The server sends the column list in its own `event: schema` frame — before the first row, and again whenever the list drifts on the **live** path (with one exception after a gap-fill, below) — and each row as a JSON array; the SDK keeps the announced list and zips every row against it, so this shape is unchanged and nothing in your code moves. It matters in two places. The row object has a **null prototype**: a ClickHouse column may legitimately be named `__proto__`, and on an ordinary object that assignment hits the inherited setter and the value disappears — so the SDK builds each row with `Object.create(null)`. Property access, spreading, `JSON.stringify` and destructuring all behave normally; what does not is anything inherited from `Object.prototype`, so use `Object.hasOwn(row, "x")` rather than `row.hasOwnProperty("x")`, and don't rely on `` `${row}` `` or `row.constructor`. (`liveQuery`'s REST backfill half still yields ordinary objects.) And a **raw** SSE consumer (a hand-rolled `EventSource`) must do the zipping itself — see [the wire format](/api#get-v1stream--server-sent-events-stream). A column the producer omitted is no longer `null` on the wire: WaveHouse's ingest validation runs ClickHouse's own parser in-process, which evaluates the column's `DEFAULT` (or its implicit zero value) before the row is published — the same as a native `INSERT` naming fewer columns than the table has. -Row values of top-level `DateTime`/`DateTime64` columns inside `data` (not timestamps nested in `Array`/`Map`/`Tuple` columns) arrive as ClickHouse itself renders them — `"2026-06-21 04:00:00.123"`, space-separated, no `Z` suffix, in the column's declared zone else the server's default — matching what `/v1/query` returns for the same row byte-for-byte, by construction (see [Timestamp rendering](/api#timestamp-rendering)). `new Date(value)` does **not** parse this form correctly out of the box: ECMAScript's `Date` constructor reads a bare `YYYY-MM-DD HH:MM:SS` string as *local* time, not UTC, so replace the space with `T` and append the zone before parsing, or parse it with a zone-aware library. +Row values of top-level `DateTime`/`DateTime64` columns inside `data` (not timestamps nested in `Array`/`Map`/`Tuple` columns) arrive as RFC 3339 in UTC, whatever zone the column declares — `"2026-06-21T04:00:00.123Z"` — matching what `/v1/query` returns for the same row byte-for-byte, by construction (see [Timestamp rendering](/api#timestamp-rendering)). `new Date(value)` parses it directly, and because the stream's `timestamp` field and a timestamp column use the same form, the SDK's comparisons between them (the live-query dedupe below, client-side `.where()` on a timestamp column) are comparisons between like spellings. ### Transport Behavior diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index f5ad93ff..9d369547 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -103,20 +103,20 @@ The tenant tunables. Every key is required (a missing one is a validation error) | Key | Seed | Description | | --- | ---- | ----------- | -| `clickhouse.addr` | `localhost:9000` | Native-protocol `host:port` — schema discovery, structured queries, pipes, `/readyz`. See [ClickHouse](#clickhouse). | -| `clickhouse.http_port` | `8123` | HTTP interface port on the same host (ingest `INSERT`s and the raw-SQL proxy), `1–65535`. | +| `clickhouse.addr` | `localhost:9000` | Native-protocol `host:port` — schema discovery and `/readyz`; queries and inserts go over the HTTP interface. See [ClickHouse](#clickhouse). | +| `clickhouse.http_port` | `8123` | HTTP interface port on the same host (structured queries, pipes, ingest `INSERT`s and the raw-SQL proxy), `1–65535`. | | `clickhouse.http_scheme` | `http` | `http` or `https` for that HTTP hop — one of the two *outbound* TLS switches, with `tls.enabled` for the native hop; unrelated to your clients' TLS. | | `clickhouse.database` | `default` | Database tables are discovered from. | | `clickhouse.username` | `default` | Connection user; the password is boot config (`WH_CH_PASSWORD`). | -| `clickhouse.query_timeout` | `30` | Seconds (`>= 1`) a ClickHouse call may take on the query paths — structured queries, pipes (a write pipe included) and `/v1/ops/query`. On `/v1/query` under a role's `max_execution_time`, the smaller of the two is sent to ClickHouse as `max_execution_time`; otherwise, on the native paths, it bounds the client deadline, from which the driver derives a server-side `max_execution_time`; on `/v1/ops/query` it is the HTTP request's deadline. | +| `clickhouse.query_timeout` | `30` | Seconds (`>= 1`) a ClickHouse call may take on the query paths — structured queries, pipes (a write pipe included) and `/v1/ops/query`. On `/v1/query` and pipes it is always sent to ClickHouse as `max_execution_time` (the smaller of it and a role's own cap), and WaveHouse's client deadline is two seconds past that, so ClickHouse names the limit that stopped a query; on `/v1/ops/query` it is the HTTP request's deadline. | | `clickhouse.tls.enabled` | `false` | Switches the native-protocol hop (`addr`) to TLS. The HTTP hop's switch stays `http_scheme`; the rest of the `tls` block applies to whichever hop uses TLS. See [ClickHouse](#clickhouse). | | `clickhouse.tls.ca_file` | `""` | PEM bundle the server certificate is verified against; empty uses the system roots. A path, read when the connection is built and re-read when the `tls` block changes — validation does not open it. | | `clickhouse.tls.cert_file` | `""` | Client certificate for mutual TLS, PEM; set together with `key_file` or not at all. | | `clickhouse.tls.key_file` | `""` | The client certificate's private key, PEM. | | `clickhouse.tls.insecure_skip_verify` | `false` | Skips server certificate verification. `true` validates with a warning: both hops then accept any certificate. | | `clickhouse.tls.server_name` | `""` | Name the server certificate is verified against; empty derives it from the host in `addr` (native) or the URL (HTTP). | -| `clickhouse.headers` | `{}` | Extra request headers for the HTTP interface (ingest `INSERT`s, `POST /v1/ops/query`). WaveHouse's own `Content-Type` and credential headers are set after them and win; naming `X-ClickHouse-User`, `X-ClickHouse-Key` or `Authorization` is a validation error, and so are two spellings of one name (names are case-insensitive). See [ClickHouse](#clickhouse). | -| `clickhouse.max_open_conns` | `10` | Native connection pool size (`>= max_idle_conns`); tenants sharing a pool size it to the largest ask among them. The open pools together must not exceed the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) when one is set. See [ClickHouse](#clickhouse). | +| `clickhouse.headers` | `{}` | Extra request headers for the HTTP interface (structured queries, pipes, ingest `INSERT`s, `POST /v1/ops/query`). WaveHouse's own `Content-Type` and credential headers are set after them and win; naming `X-ClickHouse-User`, `X-ClickHouse-Key` or `Authorization` is a validation error, and so are two spellings of one name (names are case-insensitive). See [ClickHouse](#clickhouse). | +| `clickhouse.max_open_conns` | `10` | Native connection pool size (`>= max_idle_conns`); tenants sharing a pool size it to the largest ask among them. It also caps the HTTP reader connections behind structured queries and pipes, per pool identity (URL, user, database and TLS settings). The open pools together must not exceed the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) when one is set. See [ClickHouse](#clickhouse). | | `clickhouse.max_idle_conns` | `5` | Idle native connections kept open (`>= 1`). | | `auth.jwks_url` | `""` | JWKS endpoint (absolute `http(s)` URL). When set, JWKS is the **sole** verifier and `jwt_secret` is ignored. See [Authentication](#authentication). | | `auth.role_claim` | `role` | Dot-separated JWT claim path the role is read from (e.g. `app_metadata.role`). | diff --git a/docs/src/content/docs/why-wavehouse.md b/docs/src/content/docs/why-wavehouse.md index ba08cd60..193f6f24 100644 --- a/docs/src/content/docs/why-wavehouse.md +++ b/docs/src/content/docs/why-wavehouse.md @@ -136,7 +136,7 @@ flowchart TB ```mermaid flowchart TB - subgraph single["WAVEHOUSE: 1 BINARY + CLICKHOUSE"] + subgraph single["WAVEHOUSE: 1 PROCESS + CLICKHOUSE"] direction TB Cs2["Clients"]:::neutral Cs2 <--> WHone["WaveHouse
embedded NATS · cache · auth ·
streaming · DLQ · dedup"]:::wh @@ -186,7 +186,7 @@ Tinybird wins on "zero ops to start." WaveHouse wins on "own your data plane and | Concern | Direct ClickHouse | Kafka + ClickHouse (DIY) | Tinybird | **WaveHouse** | | ------- | ----------------- | ----------------------- | -------- | ------------- | -| Single-binary deployment | — | — | N/A (SaaS) | ✓ | +| Single-process deployment | — | — | N/A (SaaS) | ✓ | | Self-hosted | ✓ | ✓ | ✗ | ✓ | | Handles N-row inserts safely | ✗ merge blowup | ✓ via Kafka | ✓ | ✓ native | | Schema validation at the edge | ✗ | Custom | ✓ | ✓ (discovers schema) | From 5aa1b81401941b2f4449391e39a277b931436282 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:49:51 -0400 Subject: [PATCH 31/70] test(typelayer): move the test engine helpers into typelayertest TestEngine, SkipWithoutArtifact and TestServerVersion lived in a non-test file of internal/typelayer, which the API process links, so the testing package (and its flags) rode into the production binary. They now live in internal/typelayer/typelayertest, beside the repo's other *test helper packages; the engine exposes TenantCause for them. The package's own tests keep a small in-package copy, since they cannot import a package that imports them, and a test pins that typelayer itself never imports testing. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- .github/workflows/README.md | 2 +- .github/workflows/ci.yml | 2 +- internal/api/ingest_test.go | 6 +- internal/api/ingest_unavailable_test.go | 8 +-- internal/app/app_test.go | 4 +- internal/stream/hub_test.go | 4 +- internal/stream/roweval_test.go | 9 +-- internal/typelayer/checks_test.go | 12 ++-- internal/typelayer/filter_test.go | 20 +++---- internal/typelayer/helpers_test.go | 58 +++++++++++++++++++ internal/typelayer/ingest_test.go | 6 +- internal/typelayer/roletable_test.go | 24 ++++---- internal/typelayer/tenancy_test.go | 52 ++++++++--------- internal/typelayer/typelayer.go | 7 ++- internal/typelayer/typelayer_test.go | 48 +++++++-------- .../typelayertest.go} | 17 ++++-- internal/typelayer/zone_test.go | 32 +++++----- 17 files changed, 189 insertions(+), 122 deletions(-) create mode 100644 internal/typelayer/helpers_test.go rename internal/typelayer/{testing.go => typelayertest/typelayertest.go} (80%) diff --git a/.github/workflows/README.md b/.github/workflows/README.md index 83d44f73..fb31013f 100644 --- a/.github/workflows/README.md +++ b/.github/workflows/README.md @@ -124,7 +124,7 @@ Never add a per-job copy of content that is a pure function of a lockfile — ke ## chtypes artifacts -`unit`, `integration` and `e2e` link the chtypes SDK (cgo dlopen of a per-ClickHouse-version `.so`/`.dylib`) and need the artifact for the line the test suite dials — today ClickHouse 26.8, matching `tests/integration/setup_test.go`'s pinned container. `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (job-level `env:` on all three) makes `typelayer.TestEngine` `t.Fatal()` if the artifact is missing instead of `t.Skip()`ing — CI must never quietly skip chtypes-backed tests. +`unit`, `integration` and `e2e` link the chtypes SDK (cgo dlopen of a per-ClickHouse-version `.so`/`.dylib`) and need the artifact for the line the test suite dials — today ClickHouse 26.8, matching `tests/integration/setup_test.go`'s pinned container. `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (job-level `env:` on all three) makes `typelayertest.TestEngine` `t.Fatal()` if the artifact is missing instead of `t.Skip()`ing — CI must never quietly skip chtypes-backed tests. `chtypes.lock` (repo root) pins the exact file + sha256 per platform/line; `scripts/fetch-chtypes.sh` wraps the SDK's own CLI with `--frozen --lock chtypes.lock`, so a fetch here can only install what the lock names, never the rolling `artifacts` release. `setup-env`'s `chtypes: "true"` input (see [Cache inventory](#cache-inventory)) restores `~/.cache/chtypes/artifacts/abi6` and always re-runs the fetch script afterward — cheap on a hit (a manifest check, not a re-download) and what turns a restored-but-unverified cache entry back into a hash-checked one every run. A lock is specific to the SDK's ABI revision: a build from another revision is never selected, so after an SDK bump that changes the revision (6 at v0.5.2) `--frozen` fails with `CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED` until the lock is regenerated the same way, and the `abi6` path and key prefix in `setup-env` move with it. Widening the pinned line set is a two-step: `scripts/fetch-chtypes.sh ` locally to update `chtypes.lock`, then add the line to `LOCK_LINES` in that script. Two hash mismatches are possible and they behave differently — don't read one as the other. **Upstream republished a pinned line under a new sha256**: `--frozen` refuses the artifact the lock does not name, and both `Dockerfile.goreleaser`'s fetch and CI's cache-miss fetch fail, until `chtypes.lock` is regenerated per platform (`go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch 26.8 --lock chtypes.lock --platform `, once each for `darwin-arm64`, `linux-amd64`, `linux-arm64`, without `--frozen`) — and `goreleaser-validate.yml`'s image job, which exercises this same fetch on every PR touching `chtypes.lock` or the release workflows, is what surfaces a republish at PR time rather than mid-release. **The cache holds a library the current lock no longer names** (a re-lock landed, so the exact key missed and `restore-keys` handed back the previous generation): this does *not* fail — measured, the CLI reports `is present but hashes … (want …) — replacing` and re-downloads, then the post-job save mints the new generation. So a re-lock costs one cold fetch per Go job on the first run and nothing after. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 98898b8d..dd663b7a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -198,7 +198,7 @@ jobs: if: needs.changes.outputs.code == 'true' runs-on: ubuntu-latest timeout-minutes: 20 - # WAVEHOUSE_TEST_REQUIRE_CHTYPES=1: typelayer.TestEngine t.Fatal()s instead of + # WAVEHOUSE_TEST_REQUIRE_CHTYPES=1: typelayertest.TestEngine t.Fatal()s instead of # t.Skip()ing when the pinned artifact isn't installed — CI must never # silently skip the chtypes-backed tests it has the artifact for. env: diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index 050bb690..ff8a111a 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -26,7 +26,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" - "github.com/Wave-RF/WaveHouse/internal/typelayer" + "github.com/Wave-RF/WaveHouse/internal/typelayer/typelayertest" "github.com/golang-jwt/jwt/v5" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -55,7 +55,7 @@ func testRegistry(t testing.TB) *discovery.SchemaRegistry { func newTestIngestHandler(t testing.TB, reg *discovery.SchemaRegistry, pub mq.Publisher) *IngestHandler { t.Helper() h := NewIngestHandler(fixedRegistry(reg), pub) - h.Types = typelayer.TestEngine(t, reg.List()...) + h.Types = typelayertest.TestEngine(t, reg.List()...) return h } @@ -64,7 +64,7 @@ func newTestIngestHandler(t testing.TB, reg *discovery.SchemaRegistry, pub mq.Pu // tenant.Default. func bindTenants(h *IngestHandler, reg *discovery.SchemaRegistry, ids ...tenant.ID) { for _, id := range ids { - h.Types.Bind(id, typelayer.TestServerVersion, "UTC", reg.List()) + h.Types.Bind(id, typelayertest.TestServerVersion, "UTC", reg.List()) } } diff --git a/internal/api/ingest_unavailable_test.go b/internal/api/ingest_unavailable_test.go index 4c6189f4..1b9dab02 100644 --- a/internal/api/ingest_unavailable_test.go +++ b/internal/api/ingest_unavailable_test.go @@ -18,7 +18,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" - "github.com/Wave-RF/WaveHouse/internal/typelayer" + "github.com/Wave-RF/WaveHouse/internal/typelayer/typelayertest" ) // The type layer judges every record, so there is no Go-side comparison left @@ -61,7 +61,7 @@ func TestIngest_UndiscoveredTable_Unavailable(t *testing.T) { buf := logtest.Capture(t, slog.LevelError) pub := &testutil.MockPublisher{} h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) - h.Types = typelayer.TestEngine(t) // bound to NO tables + h.Types = typelayertest.TestEngine(t) // bound to NO tables w := httptest.NewRecorder() h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/home"}))) @@ -82,8 +82,8 @@ func TestIngest_UnboundTenant_RefusedBeforeTheBodyIsRead(t *testing.T) { reg := testRegistry(t) pub := &testutil.MockPublisher{} h := NewIngestHandler(fixedRegistry(reg), pub) - h.Types = typelayer.TestEngine(t) // tenant.Default only, and no tables - h.Types.Bind("globex", typelayer.TestServerVersion, "UTC", reg.List()) + h.Types = typelayertest.TestEngine(t) // tenant.Default only, and no tables + h.Types.Bind("globex", typelayertest.TestServerVersion, "UTC", reg.List()) acme, ok := tenants.For("acme") require.True(t, ok) diff --git a/internal/app/app_test.go b/internal/app/app_test.go index 57338709..dfa06250 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -40,7 +40,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/testutil" "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" "github.com/Wave-RF/WaveHouse/internal/testutil/storedir" - "github.com/Wave-RF/WaveHouse/internal/typelayer" + "github.com/Wave-RF/WaveHouse/internal/typelayer/typelayertest" ) // None of these tests run in parallel: New installs a process-wide default @@ -148,7 +148,7 @@ func newApp(t *testing.T, cfg *config.Config, opts Options) *App { func newForTest(ctx context.Context, t *testing.T, opts Options) (*App, error) { t.Helper() if opts.Config != nil && opts.Config.Has(config.RoleAPI) { - typelayer.SkipWithoutArtifact(t) + typelayertest.SkipWithoutArtifact(t) } return New(ctx, opts) } diff --git a/internal/stream/hub_test.go b/internal/stream/hub_test.go index 46a70dc0..9fe46acf 100644 --- a/internal/stream/hub_test.go +++ b/internal/stream/hub_test.go @@ -22,7 +22,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" - "github.com/Wave-RF/WaveHouse/internal/typelayer" + "github.com/Wave-RF/WaveHouse/internal/typelayer/typelayertest" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" "go.opentelemetry.io/otel" @@ -498,7 +498,7 @@ func col(name, chType string) discovery.Column { func chtypesHub(tb testing.TB, store PolicySource, metric *Metrics, tables ...*discovery.TableSchema) *Hub { tb.Helper() hub := NewHub(store, nil, metric) - hub.RowEvaluator = NewRowEvaluator(typelayer.TestEngine(tb, tables...)) + hub.RowEvaluator = NewRowEvaluator(typelayertest.TestEngine(tb, tables...)) return hub } diff --git a/internal/stream/roweval_test.go b/internal/stream/roweval_test.go index c53c919a..43eaa9b4 100644 --- a/internal/stream/roweval_test.go +++ b/internal/stream/roweval_test.go @@ -13,6 +13,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/typelayer" + "github.com/Wave-RF/WaveHouse/internal/typelayer/typelayertest" ) // recordingEvaluator answers every row the same way and counts what the Hub @@ -268,8 +269,8 @@ func TestWithheldReason_ClassifiesTypeLayerErrors(t *testing.T) { // judged normally. func TestEngineEvaluator_TenantsAreIndependent(t *testing.T) { t.Parallel() - eng := typelayer.TestEngine(t, clicksTable()) - eng.Bind("acme", typelayer.TestServerVersion, "UTC", []*discovery.TableSchema{clicksTable()}) + eng := typelayertest.TestEngine(t, clicksTable()) + eng.Bind("acme", typelayertest.TestServerVersion, "UTC", []*discovery.TableSchema{clicksTable()}) eval := NewRowEvaluator(eng) cols := []string{"page", "secret", "tenant_id"} row := json.RawMessage(`["/a","x","acme"]`) @@ -376,7 +377,7 @@ func TestHub_RowFilter_FilterOnAbsentColumnWithholds(t *testing.T) { Filter: map[string]policy.Filter{"region": {Eq: new("eu")}}, }}}, }} - eng := typelayer.TestEngine(t, roleWidthTable()) + eng := typelayertest.TestEngine(t, roleWidthTable()) hub := NewHub(staticPolicy(p), nil, nil) hub.RowEvaluator = NewRowEvaluator(eng) sub := NewSubscriber(nil, nil) @@ -408,7 +409,7 @@ func TestHub_RowFilter_UnknownColumnIsDrift(t *testing.T) { t.Parallel() p := rowFilterPolicy() p.Tables["clicks"]["public"] = policy.RolePermissions{Select: &policy.SelectPermissions{}} - eng := typelayer.TestEngine(t, roleWidthTable()) + eng := typelayertest.TestEngine(t, roleWidthTable()) hub := NewHub(staticPolicy(p), nil, nil) hub.RowEvaluator = NewRowEvaluator(eng) acme := NewSubscriber(map[string]any{"tenant": "acme"}, nil) diff --git a/internal/typelayer/checks_test.go b/internal/typelayer/checks_test.go index e732dd11..cff3127b 100644 --- a/internal/typelayer/checks_test.go +++ b/internal/typelayer/checks_test.go @@ -35,7 +35,7 @@ const checksBody = `{"id":1,"tenant":"acme","kind":"a"}` + "\n" + func checksHandle(t *testing.T) *Table { t.Helper() - eng := TestEngine(t, checksTable()) + eng := testEngine(t, checksTable()) tbl, err := eng.Table(tenant.Default, "checks") require.NoError(t, err) t.Cleanup(tbl.Release) @@ -155,7 +155,7 @@ func TestIngestChecks_FailsClosed(t *testing.T) { "an integer claim that does not fit the column is 'the data says no' (403)") // On any other column the server's own reader throws. - eng := TestEngine(t, &discovery.TableSchema{Name: "ratios", Columns: []discovery.Column{ + eng := testEngine(t, &discovery.TableSchema{Name: "ratios", Columns: []discovery.Column{ {Name: "id", Type: "UInt32", Position: 1}, {Name: "ratio", Type: "Float64", Position: 2}, }}) @@ -193,7 +193,7 @@ func TestIngestChecks_FailsClosed(t *testing.T) { // from the injected DEFAULT, which the compiler may itself wrap. A claim that // fits admits exactly as before. func TestIngestChecks_IntegerClaimThatDoesNotFitIsRefused(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) const over = "18446744073709551621" // 2^64+5 body := []byte(`{"id":1}` + "\n" + `{"id":2,"amount":5}` + "\n" + `{"id":3,"amount":6}` + "\n") @@ -332,8 +332,8 @@ func TestIngestChecks_OnTheWithNamesFormats(t *testing.T) { } func TestIngest_UnavailableTable(t *testing.T) { - eng := TestEngine(t, checksTable()) - eng.Bind(tenant.Default, TestServerVersion, "UTC", nil) + eng := testEngine(t, checksTable()) + eng.Bind(tenant.Default, testServerVersion, "UTC", nil) _, err := eng.Table(tenant.Default, "checks") require.Error(t, err) @@ -461,7 +461,7 @@ func TestIngest_WithNamesFormats(t *testing.T) { // Only when parallel-shared is ~GOMAXPROCS× worse than serial does a pool have // anything to win, and parallel-pooled is then the size of the win. func BenchmarkIngest_HandlePool(b *testing.B) { - eng := TestEngine(b, checksTable()) + eng := testEngine(b, checksTable()) tbl, err := eng.Table(tenant.Default, "checks") require.NoError(b, err) defer tbl.Release() diff --git a/internal/typelayer/filter_test.go b/internal/typelayer/filter_test.go index 8383e5ab..8639ab6f 100644 --- a/internal/typelayer/filter_test.go +++ b/internal/typelayer/filter_test.go @@ -38,7 +38,7 @@ const sampleRow = `[7, "acme", "2026-01-15 10:30:00", "12.50", -5, 0.1, ["a"]]` func parsedRow(t *testing.T) (*Table, *Row) { t.Helper() - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) t.Cleanup(tbl.Release) @@ -175,7 +175,7 @@ func intDomain(typ string) (*big.Int, *big.Int) { // is the canonical spelling of a value the column can hold, and false // otherwise — never an over-admit, and never a thrown row. func TestVisible_IntegerClaimsMatchExactlyWhatFits(t *testing.T) { - eng := TestEngine(t, intsTable()) + eng := testEngine(t, intsTable()) tbl, err := eng.Table(tenant.Default, "ints") require.NoError(t, err) t.Cleanup(tbl.Release) @@ -296,7 +296,7 @@ func storedRow(t *testing.T, tbl *Table, tenant string) *Row { // for all of them, so the two read surfaces disagreed. With the encoding they // agree. func TestVisible_EscapedStringParamsMatchTheStoredValue(t *testing.T) { - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) t.Cleanup(tbl.Release) @@ -430,7 +430,7 @@ func TestVisibleWithReason_LabelsTheCause(t *testing.T) { // refusal and every predicate over it withholds (measured on the 26.6 // artifact: ParseBlock reports no call-level error for a malformed row). func TestParseRow_ColumnsDriftIsAnError(t *testing.T) { - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) defer tbl.Release() @@ -477,7 +477,7 @@ func defaultsTable() *discovery.TableSchema { // full-width row would have stored as NULL — and a DEFAULT over a listed // column is computed from the listed value. func TestParseRow_ColumnSubsetTakesTheServersDefaults(t *testing.T) { - eng := TestEngine(t, defaultsTable()) + eng := testEngine(t, defaultsTable()) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) defer tbl.Release() @@ -513,7 +513,7 @@ func TestParseRow_ColumnSubsetTakesTheServersDefaults(t *testing.T) { } func TestParseRow_AcceptsALineWithOrWithoutNewline(t *testing.T) { - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) defer tbl.Release() @@ -576,7 +576,7 @@ func TestFilterCache_BoundedUnderTenantValueChurn(t *testing.T) { // makes the cache not a DoS, and it must be a bound on the TABLE — a pool of // handles must not multiply it. func TestFilterCache_BudgetIsSplitAcrossTheHandlePool(t *testing.T) { - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) defer tbl.Release() @@ -595,7 +595,7 @@ func TestFilterCache_BudgetIsSplitAcrossTheHandlePool(t *testing.T) { // TestFilterCache_GenerationInvalidatesEntries: a filter only answers for the // schema handle it was compiled against, so a rebind must not reuse one. func TestFilterCache_GenerationInvalidatesEntries(t *testing.T) { - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) visible := func() bool { tbl, err := eng.Table(tenant.Default, "rows") @@ -612,7 +612,7 @@ func TestFilterCache_GenerationInvalidatesEntries(t *testing.T) { // over it are rebuilt, while the wire arity stays the same. changed := rowsTable() changed.Columns[0].Type = "UInt16" - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{changed}) assert.True(t, visible(), "a rebind recompiles rather than reusing a freed handle") } @@ -645,7 +645,7 @@ func TestVisible_ConcurrentSubscribers(t *testing.T) { // safe; a slot chosen per call rather than per Row would show up here as a // filter and a block on different handles. func TestTable_ConcurrentAcrossThePool(t *testing.T) { - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) const goroutines = 16 var wg sync.WaitGroup diff --git a/internal/typelayer/helpers_test.go b/internal/typelayer/helpers_test.go new file mode 100644 index 00000000..50670270 --- /dev/null +++ b/internal/typelayer/helpers_test.go @@ -0,0 +1,58 @@ +package typelayer + +import ( + "go/build" + "os" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// This package's own tests cannot import typelayertest (it imports this +// package), so they carry the same three pieces: the version they bind, its +// line, and the engine helper. Kept in step with typelayertest.TestEngine. +const ( + testServerVersion = "26.8.15.10" + testLine = "26.8" +) + +// testEngine opens an Engine on the SDK's default search path and binds +// tables for tenant.Default at testServerVersion in UTC, skipping the test +// when the artifact is absent or does not load — failing it under +// WAVEHOUSE_TEST_REQUIRE_CHTYPES=1. +func testEngine(t testing.TB, tables ...*discovery.TableSchema) *Engine { + t.Helper() + eng, err := NewEngine(Config{}) + if err != nil { + skipWithoutArtifact(t, err.Error()) + } + t.Cleanup(eng.Close) + eng.Bind(tenant.Default, testServerVersion, "UTC", tables) + if cause := eng.TenantCause(tenant.Default); cause != "" { + skipWithoutArtifact(t, cause) + } + return eng +} + +// TestPackageDoesNotImportTesting: the API process links this package, so a +// test helper in a non-test file drags the testing package (and its flags) +// into the production binary. Test helpers for other packages live in +// typelayertest. +func TestPackageDoesNotImportTesting(t *testing.T) { + pkg, err := build.ImportDir(".", 0) + require.NoError(t, err) + assert.NotContains(t, pkg.Imports, "testing") +} + +func skipWithoutArtifact(t testing.TB, cause string) { + t.Helper() + const msg = "chtypes artifact for " + testLine + " not installed: run `scripts/fetch-chtypes.sh`" + if os.Getenv("WAVEHOUSE_TEST_REQUIRE_CHTYPES") == "1" { + t.Fatalf("%s\n%s", msg, cause) + } + t.Skipf("%s\n%s", msg, cause) +} diff --git a/internal/typelayer/ingest_test.go b/internal/typelayer/ingest_test.go index 5b72e213..276cd21c 100644 --- a/internal/typelayer/ingest_test.go +++ b/internal/typelayer/ingest_test.go @@ -15,7 +15,7 @@ import ( func ingestTable(t *testing.T) *Table { t.Helper() - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) t.Cleanup(tbl.Release) @@ -105,7 +105,7 @@ func TestIngest_ComputedColumnsAreRejectedPerRecord(t *testing.T) { {Name: "a", Type: "UInt8", DefaultKind: "ALIAS", DefaultExpression: "id + 2", HasDefault: true, Position: 4}, }, } - eng := TestEngine(t, schema) + eng := testEngine(t, schema) tbl, err := eng.Table(tenant.Default, "computed") require.NoError(t, err) defer tbl.Release() @@ -234,7 +234,7 @@ func gatedTable() *discovery.TableSchema { // every record is refused (455 and 44 on the 26.6 and 26.8 artifacts), which // is what this table got on every call before. func TestIngest_TypeGatedColumnsInsertAndFilter(t *testing.T) { - eng := TestEngine(t, gatedTable()) + eng := testEngine(t, gatedTable()) tbl, err := eng.Table(tenant.Default, "gated") require.NoError(t, err) defer tbl.Release() diff --git a/internal/typelayer/roletable_test.go b/internal/typelayer/roletable_test.go index dffbb46c..e19a3f11 100644 --- a/internal/typelayer/roletable_test.go +++ b/internal/typelayer/roletable_test.go @@ -37,7 +37,7 @@ func roleTableFor(t *testing.T, eng *Engine, shape RoleShape) *Table { // TestRoleTable_IdentityShapeIsTheBaseTable: a role that may write everything // and injects nothing costs no second compile and no cache entry. func TestRoleTable_IdentityShapeIsTheBaseTable(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) base, err := eng.Table(tenant.Default, "orders") require.NoError(t, err) @@ -56,7 +56,7 @@ func TestRoleTable_IdentityShapeIsTheBaseTable(t *testing.T) { // naming it is ClickHouse's own per-row code 117 rather than a Go key walk's // 403 — and the exported row carries the ROLE's column list. func TestRoleTable_DeniedColumnIsAbsentFromTheSchema(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) tbl := roleTableFor(t, eng, RoleShape{Columns: []string{"id", "tenant", "amount"}}) assert.Equal(t, []string{"id", "tenant", "amount"}, tbl.WireColumns) @@ -80,7 +80,7 @@ func TestRoleTable_DeniedColumnIsAbsentFromTheSchema(t *testing.T) { // the 26.6 and 26.8 artifacts: DEFAULT '' fills a column the record // omits, and a value the record DOES supply still wins. func TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"tenant": "acme"}}) assert.Equal(t, []string{"id", "tenant", "secret", "amount"}, tbl.WireColumns) @@ -103,7 +103,7 @@ func TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue(t *testing.T // breakout attempt is the case that matters — it must not add, remove or // retype a single column. func TestRoleTable_LiteralEscaping(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) for name, value := range map[string]string{ "apostrophe": "O'Brien", @@ -144,7 +144,7 @@ func TestRoleTable_LiteralEscaping(t *testing.T) { // cannot read is a compile refusal (ClickHouse code 6). It must be Unavailable // — a 503 — never a handle that silently drops the injection. func TestRoleTable_UnparseableLiteralFailsClosed(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Defaults: map[string]string{"amount": "abc"}}) require.Error(t, err) @@ -156,7 +156,7 @@ func TestRoleTable_UnparseableLiteralFailsClosed(t *testing.T) { // does not carry cannot be expressed. Dropping it silently would turn "force // this value" into "whatever the caller sent". func TestRoleTable_ContradictoryShapeIsAnError(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{ Columns: []string{"id", "amount"}, @@ -179,7 +179,7 @@ func TestRoleTable_DefaultIntoAComputedColumnIsRefused(t *testing.T) { Name: "double", Type: "UInt64", HasDefault: true, DefaultKind: "MATERIALIZED", DefaultExpression: "amount * 2", Position: 5, }) - eng := TestEngine(t, ts) + eng := testEngine(t, ts) _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Defaults: map[string]string{"double": "1"}}) require.Error(t, err) @@ -196,7 +196,7 @@ func TestRoleTable_DefaultIntoAComputedColumnIsRefused(t *testing.T) { // handle (a compile per request is the thing this cache exists to stop), a // different shape must not, and a rebind must invalidate both. func TestRoleTable_CachedPerShapeAndGeneration(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) shape := RoleShape{Columns: []string{"tenant", "id", "amount"}, Defaults: map[string]string{"tenant": "acme"}} role := func(s RoleShape) *Table { @@ -228,7 +228,7 @@ func TestRoleTable_CachedPerShapeAndGeneration(t *testing.T) { // A rebind closes every projection; the next lookup compiles a fresh one. changed := ordersTable() changed.Columns[3].Type = "UInt32" - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{changed}) after := role(shape) assert.NotSame(t, first, after) @@ -238,7 +238,7 @@ func TestRoleTable_CachedPerShapeAndGeneration(t *testing.T) { // TestRoleTable_NegativeEntryStopsRecompiling: a shape that will not compile // costs one compile and one log line per generation, like filterCache. func TestRoleTable_NegativeEntryStopsRecompiling(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) for range 3 { _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Defaults: map[string]string{"amount": "abc"}}) @@ -254,7 +254,7 @@ func TestRoleTable_NegativeEntryStopsRecompiling(t *testing.T) { // tenant claims and are baked into compiled handles, so the cache must be // bounded exactly like filterCache. func TestRoleTable_BoundedUnderTenantValueChurn(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) base, err := eng.Table(tenant.Default, "orders") require.NoError(t, err) @@ -287,7 +287,7 @@ func TestRoleTable_BoundedUnderTenantValueChurn(t *testing.T) { // TestRoleTable_HasItsOwnHandlePool: a role shape holds its own single handle, // not the base table's pool (256 shapes x the pool would be unbounded memory). func TestRoleTable_HasItsOwnHandlePool(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"tenant": "acme"}}) assert.Len(t, tbl.pool.list(), roleHandles) assert.Equal(t, int64(roleHandles), tbl.pool.limit.Load(), "a role shape never grows past its one handle") diff --git a/internal/typelayer/tenancy_test.go b/internal/typelayer/tenancy_test.go index d6fa46e4..3212abe3 100644 --- a/internal/typelayer/tenancy_test.go +++ b/internal/typelayer/tenancy_test.go @@ -44,11 +44,11 @@ func unavailable(t *testing.T, eng *Engine, id tenant.ID) *Unavailable { // naming both zones; the first tenant, and a third in the line's own zone, // keep answering. func TestBind_TenantsOnOneLineWithDifferentZones(t *testing.T) { - eng := TestEngine(t, eventsTable()) // tenant.Default, UTC + eng := testEngine(t, eventsTable()) // tenant.Default, UTC - eng.Bind("berlin", TestServerVersion, "Europe/Berlin", []*discovery.TableSchema{eventsTable()}) - eng.Bind("utc", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) - eng.Bind("unnamed", TestServerVersion, "", []*discovery.TableSchema{eventsTable()}) + eng.Bind("berlin", testServerVersion, "Europe/Berlin", []*discovery.TableSchema{eventsTable()}) + eng.Bind("utc", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind("unnamed", testServerVersion, "", []*discovery.TableSchema{eventsTable()}) u := unavailable(t, eng, "berlin") assert.Empty(t, u.Table, "the cause covers every table of the tenant") @@ -69,12 +69,12 @@ func TestBind_TenantsOnOneLineWithDifferentZones(t *testing.T) { // TestBind_ZoneRefusalClearsWhenTheZoneMatchesAgain: the refusal is about the // zone the server reports now, not a sticky mark on the tenant. func TestBind_ZoneRefusalClearsWhenTheZoneMatchesAgain(t *testing.T) { - eng := TestEngine(t, eventsTable()) - eng.Bind("moved", TestServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) + eng := testEngine(t, eventsTable()) + eng.Bind("moved", testServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) u := unavailable(t, eng, "moved") assert.Contains(t, u.Cause, `"Asia/Tokyo"`) - eng.Bind("moved", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind("moved", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) answers(t, eng, "moved") } @@ -82,7 +82,7 @@ func TestBind_ZoneRefusalClearsWhenTheZoneMatchesAgain(t *testing.T) { // artifact covers is refused with the SDK's own message; every other tenant // answers, and the tenant answers once its line resolves. func TestBind_MissingLineIsThatTenantOnly(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) eng.Bind("old", "1.2.3.4", "UTC", []*discovery.TableSchema{eventsTable()}) u := unavailable(t, eng, "old") @@ -90,14 +90,14 @@ func TestBind_MissingLineIsThatTenantOnly(t *testing.T) { answers(t, eng, tenant.Default) - eng.Bind("old", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind("old", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) answers(t, eng, "old") } // TestTable_UnboundTenantIsUnavailable: a tenant discovery has not bound yet // is a 503, never a 404 — the table may well exist. func TestTable_UnboundTenantIsUnavailable(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) u := unavailable(t, eng, "nobody") assert.Equal(t, causeUnbound, u.Cause) } @@ -106,8 +106,8 @@ func TestTable_UnboundTenantIsUnavailable(t *testing.T) { // not share a handle (sharing is a separate decision), and one tenant's // rebind leaves the other's generation alone. func TestBind_TenantsCompileSeparateHandles(t *testing.T) { - eng := TestEngine(t, eventsTable()) - eng.Bind("other", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng := testEngine(t, eventsTable()) + eng.Bind("other", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) a, err := eng.Table(tenant.Default, "events") require.NoError(t, err) @@ -120,7 +120,7 @@ func TestBind_TenantsCompileSeparateHandles(t *testing.T) { changed := eventsTable() changed.Columns = append(changed.Columns, discovery.Column{Name: "extra", Type: "String", Position: 8}) - eng.Bind("other", TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + eng.Bind("other", testServerVersion, "UTC", []*discovery.TableSchema{changed}) a, err = eng.Table(tenant.Default, "events") require.NoError(t, err) @@ -140,8 +140,8 @@ func TestBind_TenantsCompileSeparateHandles(t *testing.T) { // that moment; its handles close once the holder lets go; and another // tenant's handles are untouched throughout. func TestForget_ClosesOnlyThatTenantAndDoesNotBlock(t *testing.T) { - eng := TestEngine(t, eventsTable()) - eng.Bind("gone", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng := testEngine(t, eventsTable()) + eng.Bind("gone", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) held, err := eng.Table("gone", "events") require.NoError(t, err) @@ -190,21 +190,21 @@ func TestForget_ClosesOnlyThatTenantAndDoesNotBlock(t *testing.T) { answers(t, eng, tenant.Default) // A later Bind brings the tenant back on fresh handles. - eng.Bind("gone", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind("gone", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) answers(t, eng, "gone") } // TestClose_WaitsForForgetAndRefusesAfterwards: shutdown leaves no teardown // running, and nothing answers or binds after it. func TestClose_WaitsForForgetAndRefusesAfterwards(t *testing.T) { - eng := TestEngine(t, eventsTable()) - eng.Bind("a", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng := testEngine(t, eventsTable()) + eng.Bind("a", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) eng.Forget("a") eng.Close() u := unavailable(t, eng, tenant.Default) assert.Equal(t, causeClosed, u.Cause) - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) u = unavailable(t, eng, tenant.Default) assert.Equal(t, causeClosed, u.Cause) eng.Forget(tenant.Default) // no-op, no panic @@ -214,7 +214,7 @@ func TestClose_WaitsForForgetAndRefusesAfterwards(t *testing.T) { // every handle busy compiles one more, up to the limit; past it, calls share // a busy handle; and an idle handle is reused rather than grown past. func TestPool_GrowsLazilyToItsLimit(t *testing.T) { - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) defer tbl.Release() @@ -257,7 +257,7 @@ func TestPool_GrowsLazilyToItsLimit(t *testing.T) { // handle rather than wait for a compile), never past its limit; holding Rows // while more arrive grows it to exactly the limit. func TestTable_PoolGrowsUnderConcurrentHolders(t *testing.T) { - eng := TestEngine(t, rowsTable()) + eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) defer tbl.Release() @@ -306,7 +306,7 @@ func TestTable_PoolGrowsUnderConcurrentHolders(t *testing.T) { // request that holds a projection and then looks the base table up is not // deadlocked against the rebind waiting for that projection. func TestBind_HeldRoleTableDoesNotBlockBaseLookups(t *testing.T) { - eng := TestEngine(t, ordersTable()) + eng := testEngine(t, ordersTable()) rt, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Columns: []string{"id", "tenant"}}) require.NoError(t, err) @@ -314,7 +314,7 @@ func TestBind_HeldRoleTableDoesNotBlockBaseLookups(t *testing.T) { changed.Columns[3].Type = "UInt32" bound := make(chan struct{}) go func() { - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{changed}) close(bound) }() @@ -346,7 +346,7 @@ func TestBind_HeldRoleTableDoesNotBlockBaseLookups(t *testing.T) { // safe; every answer is either a verdict or an Unavailable, never a panic or // a hang. func TestEngine_ConcurrentBindTableForget(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) ids := []tenant.ID{"t0", "t1", "t2", "t3"} schemas := func(i int) []*discovery.TableSchema { ts := eventsTable() @@ -363,7 +363,7 @@ func TestEngine_ConcurrentBindTableForget(t *testing.T) { id := ids[(g+i)%len(ids)] switch (g + i) % 4 { case 0, 1: - eng.Bind(id, TestServerVersion, "UTC", schemas(i)) + eng.Bind(id, testServerVersion, "UTC", schemas(i)) case 2: eng.Forget(id) default: @@ -385,7 +385,7 @@ func TestEngine_ConcurrentBindTableForget(t *testing.T) { eng.retiring.Wait() for _, id := range ids { - eng.Bind(id, TestServerVersion, "UTC", schemas(0)) + eng.Bind(id, testServerVersion, "UTC", schemas(0)) answers(t, eng, id) } answers(t, eng, tenant.Default) diff --git a/internal/typelayer/typelayer.go b/internal/typelayer/typelayer.go index ac6246cc..7b55aa8f 100644 --- a/internal/typelayer/typelayer.go +++ b/internal/typelayer/typelayer.go @@ -268,9 +268,10 @@ func (e *Engine) Close() { e.retiring.Wait() } -// tenantCause is the tenant-wide cause Table would report, "" when the tenant -// is bound and answering, for TestEngine. -func (e *Engine) tenantCause(id tenant.ID) string { +// TenantCause is the tenant-wide cause every Table of tenant id would report, +// "" when the tenant is bound and answering. A table-level cause (a table +// that did not compile) is not tenant-wide and is not reported here. +func (e *Engine) TenantCause(id tenant.ID) string { e.mu.RLock() set := e.tenants[id] e.mu.RUnlock() diff --git a/internal/typelayer/typelayer_test.go b/internal/typelayer/typelayer_test.go index 60e3a6a2..7c6d0792 100644 --- a/internal/typelayer/typelayer_test.go +++ b/internal/typelayer/typelayer_test.go @@ -34,7 +34,7 @@ func eventsTable() *discovery.TableSchema { } func TestBind_CompilesAndExposesWireColumns(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) @@ -47,7 +47,7 @@ func TestBind_CompilesAndExposesWireColumns(t *testing.T) { } func TestTable_UnknownTableIsUnavailable(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) _, err := eng.Table(tenant.Default, "nosuch") require.Error(t, err) assert.True(t, IsUnavailable(err)) @@ -58,7 +58,7 @@ func TestTable_UnknownTableIsUnavailable(t *testing.T) { // same columns must not invalidate handles or cached filters, and one that // discovers a new column must. func TestBind_GenerationBumpsOnlyOnSignatureChange(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) generation := func() uint64 { tbl, err := eng.Table(tenant.Default, "events") @@ -68,12 +68,12 @@ func TestBind_GenerationBumpsOnlyOnSignatureChange(t *testing.T) { } require.Equal(t, uint64(1), generation()) - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) assert.Equal(t, uint64(1), generation(), "identical columns keep the handle") changed := eventsTable() changed.Columns = append(changed.Columns, discovery.Column{Name: "extra", Type: "String", Position: 8}) - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{changed}) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) @@ -85,8 +85,8 @@ func TestBind_GenerationBumpsOnlyOnSignatureChange(t *testing.T) { // TestBind_DroppedTableBecomesUnavailable: a table that leaves the database // must stop answering rather than serve a handle for a schema that is gone. func TestBind_DroppedTableBecomesUnavailable(t *testing.T) { - eng := TestEngine(t, eventsTable()) - eng.Bind(tenant.Default, TestServerVersion, "UTC", nil) + eng := testEngine(t, eventsTable()) + eng.Bind(tenant.Default, testServerVersion, "UTC", nil) _, err := eng.Table(tenant.Default, "events") require.Error(t, err) @@ -97,7 +97,7 @@ func TestBind_DroppedTableBecomesUnavailable(t *testing.T) { // every directory it searched — that is the whole diagnostic, so it is passed // through verbatim. func TestBind_MissingArtifact_UnavailableWithSDKMessage(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) eng.Bind(tenant.Default, "1.2.3.4", "UTC", []*discovery.TableSchema{eventsTable()}) _, err := eng.Table(tenant.Default, "events") @@ -107,7 +107,7 @@ func TestBind_MissingArtifact_UnavailableWithSDKMessage(t *testing.T) { assert.Contains(t, err.Error(), "Looked in:") // Rebinding a version that does resolve clears the tenant's cause. - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) tbl.Release() @@ -118,9 +118,9 @@ func TestBind_MissingArtifact_UnavailableWithSDKMessage(t *testing.T) { // be adopted in-process. Every table of that tenant must stop answering, // loudly. func TestBind_TimezoneMismatchIsTenantWideAndNamesBothZones(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) - eng.Bind(tenant.Default, TestServerVersion, "Europe/Berlin", []*discovery.TableSchema{eventsTable()}) + eng.Bind(tenant.Default, testServerVersion, "Europe/Berlin", []*discovery.TableSchema{eventsTable()}) _, err := eng.Table(tenant.Default, "events") require.Error(t, err) require.True(t, IsUnavailable(err)) @@ -136,7 +136,7 @@ func TestBind_CompileRefusalIsPerTable(t *testing.T) { Name: "broken", Columns: []discovery.Column{{Name: "x", Type: "NotAType(9)", Position: 1}}, } - eng := TestEngine(t, eventsTable(), broken) + eng := testEngine(t, eventsTable(), broken) good, err := eng.Table(tenant.Default, "events") require.NoError(t, err) @@ -177,7 +177,7 @@ func renderExpr(t *testing.T, tbl *Table, preds ...Predicate) (string, map[strin } func TestRender_QuotesIdentifiersAndBindsEveryValue(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) defer tbl.Release() @@ -206,7 +206,7 @@ func TestRender_QuotesIdentifiersAndBindsEveryValue(t *testing.T) { // chsql.StrictInt expression the query path renders; every other column keeps // the plain {pN:String} form. func TestRender_IntegerColumnsBindThroughTheStrictCast(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) defer tbl.Release() @@ -243,7 +243,7 @@ func TestRender_QuotesEveryIdentifier(t *testing.T) { Name: "reserved", Columns: []discovery.Column{{Name: "all", Type: "String", Position: 1}}, } - eng := TestEngine(t, reserved) + eng := testEngine(t, reserved) tbl, err := eng.Table(tenant.Default, "reserved") require.NoError(t, err) defer tbl.Release() @@ -257,7 +257,7 @@ func TestRender_BackticksAnIdentifierThatNeedsIt(t *testing.T) { Name: "odd", Columns: []discovery.Column{{Name: "weird name", Type: "String", Position: 1}}, } - eng := TestEngine(t, odd) + eng := testEngine(t, odd) tbl, err := eng.Table(tenant.Default, "odd") require.NoError(t, err) defer tbl.Release() @@ -267,7 +267,7 @@ func TestRender_BackticksAnIdentifierThatNeedsIt(t *testing.T) { } func TestRender_RefusesWhatItCannotExpress(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) defer tbl.Release() @@ -312,7 +312,7 @@ func TestBind_WireColumnsComeFromTheCompiledHandle(t *testing.T) { {Name: "eph", Type: "UInt8", DefaultKind: "EPHEMERAL", Position: 5}, }, } - eng := TestEngine(t, kinds) + eng := testEngine(t, kinds) tbl, err := eng.Table(tenant.Default, "kinds") require.NoError(t, err) defer tbl.Release() @@ -331,7 +331,7 @@ func TestBind_WireColumnsComeFromTheCompiledHandle(t *testing.T) { // TestBind_HandlePoolPerTable: every table gets a pool of one handle, and a // rebind replaces all of it. func TestBind_HandlePoolPerTable(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) tbl, err := eng.Table(tenant.Default, "events") require.NoError(t, err) assert.Len(t, tbl.pool.list(), 1) @@ -340,7 +340,7 @@ func TestBind_HandlePoolPerTable(t *testing.T) { changed := eventsTable() changed.Columns[0].Type = "UInt16" - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{changed}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{changed}) tbl, err = eng.Table(tenant.Default, "events") require.NoError(t, err) @@ -364,7 +364,7 @@ func compiledColumnNames(schema *chtypes.LoadedSchema) []string { // library to ask) must spell every name byte for byte alike, so a divergence // fails here rather than in a customer's query. func TestQuoteIdentifier_AgreesWithChsql(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) lib := boundLib(eng, tenant.Default) require.NotNil(t, lib) @@ -387,7 +387,7 @@ func TestQuoteIdentifier_AgreesWithChsql(t *testing.T) { // artifact that cannot load is not a boot failure; the first Bind for its line // reports the SDK's own error as that tenant's Unavailable. func TestNewEngine_OpensNoLibraryAtConstruction(t *testing.T) { - TestEngine(t) // skips (or fails under WAVEHOUSE_TEST_REQUIRE_CHTYPES) without the real artifact + testEngine(t) // skips (or fails under WAVEHOUSE_TEST_REQUIRE_CHTYPES) without the real artifact // The explicit directory is searched first, so its broken copy of the test // line shadows the working one in the per-user cache. @@ -396,13 +396,13 @@ func TestNewEngine_OpensNoLibraryAtConstruction(t *testing.T) { require.NoError(t, os.MkdirAll(line, 0o750)) require.NoError(t, os.WriteFile(filepath.Join(line, "manifest.json"), fmt.Appendf(nil, `{"library":"libchtypes.so","clickhouse_version":%q,"clickhouse_minor":%q}`, - TestServerVersion+"-lts", testLine), 0o600)) + testServerVersion+"-lts", testLine), 0o600)) eng, err := NewEngine(Config{RegistryDir: dir}) require.NoError(t, err, "a broken artifact is not a construction error") t.Cleanup(eng.Close) - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) tbl, err := eng.Table(tenant.Default, "events") if err == nil { tbl.Release() // so the failure below is reported instead of Close deadlocking on it diff --git a/internal/typelayer/testing.go b/internal/typelayer/typelayertest/typelayertest.go similarity index 80% rename from internal/typelayer/testing.go rename to internal/typelayer/typelayertest/typelayertest.go index 56a98ea1..6ca4b539 100644 --- a/internal/typelayer/testing.go +++ b/internal/typelayer/typelayertest/typelayertest.go @@ -1,4 +1,8 @@ -package typelayer +// Package typelayertest opens a typelayer.Engine on the locked chtypes +// artifact for tests in other packages, and skips a test that needs one when +// it is not installed. It is test-only: nothing the wavehouse binary links +// imports it, so the testing package stays out of production builds. +package typelayertest import ( "fmt" @@ -12,6 +16,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) // TestServerVersion is the ClickHouse version TestEngine binds to — the patch @@ -36,7 +41,7 @@ const missingArtifact = "chtypes artifact for " + testLine + " (ABI revision 6) // line is not on the SDK's search path — or fails it when // WAVEHOUSE_TEST_REQUIRE_CHTYPES=1. It opens no library. A test that boots // anything constructing an Engine (the API process role) calls it first, since -// NewEngine refuses to start without an artifact. +// typelayer.NewEngine refuses to start without an artifact. func SkipWithoutArtifact(t testing.TB) { t.Helper() if cause := artifactMissing(); cause != "" { @@ -70,22 +75,24 @@ func artifactMissing() string { // given tables for tenant.Default at TestServerVersion in UTC. It skips the // test when the artifact is absent or does not load (fails it under // WAVEHOUSE_TEST_REQUIRE_CHTYPES=1), and closes the Engine when the test ends. -func TestEngine(t testing.TB, tables ...*discovery.TableSchema) *Engine { +func TestEngine(t testing.TB, tables ...*discovery.TableSchema) *typelayer.Engine { t.Helper() SkipWithoutArtifact(t) - eng, err := NewEngine(Config{}) + eng, err := typelayer.NewEngine(typelayer.Config{}) if err != nil { skipUnlessRequired(t, err.Error()) } t.Cleanup(eng.Close) eng.Bind(tenant.Default, TestServerVersion, "UTC", tables) - if cause := eng.tenantCause(tenant.Default); cause != "" { + if cause := eng.TenantCause(tenant.Default); cause != "" { skipUnlessRequired(t, cause) } return eng } +// skipUnlessRequired skips t because the artifact cannot serve it, naming the command that +// installs it — or fails t under WAVEHOUSE_TEST_REQUIRE_CHTYPES=1. func skipUnlessRequired(t testing.TB, cause string) { t.Helper() if os.Getenv(RequireEnv) == "1" { diff --git a/internal/typelayer/zone_test.go b/internal/typelayer/zone_test.go index a1a3cceb..9039659a 100644 --- a/internal/typelayer/zone_test.go +++ b/internal/typelayer/zone_test.go @@ -53,12 +53,12 @@ func withMsg(recs []map[string]any, msg string) []map[string]any { // own check gives, with the SDK's words kept, and the first Engine's tenant // keeps answering. func TestBind_SDKZoneRefusalIsThatTenantsUnavailable(t *testing.T) { - first := TestEngine(t, eventsTable()) // the line is open in UTC from here on + first := testEngine(t, eventsTable()) // the line is open in UTC from here on second, err := NewEngine(Config{}) require.NoError(t, err) t.Cleanup(second.Close) - second.Bind("tokyo", TestServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) + second.Bind("tokyo", testServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) u := unavailable(t, second, "tokyo") assert.Empty(t, u.Table) @@ -69,7 +69,7 @@ func TestBind_SDKZoneRefusalIsThatTenantsUnavailable(t *testing.T) { // The refusal is the tenant's, not the registry's: the same Engine serves a // tenant in the line's zone. - second.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + second.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) answers(t, second, tenant.Default) answers(t, first, tenant.Default) } @@ -82,7 +82,7 @@ func TestBind_SDKZoneRefusalIsThatTenantsUnavailable(t *testing.T) { // installed library, the same image is refused in another zone and served in // its own. func TestBind_UnstatableLibraryIsThatTenantsUnavailable(t *testing.T) { - first := TestEngine(t, eventsTable()) // the line is open in UTC from here on + first := testEngine(t, eventsTable()) // the line is open in UTC from here on installed := boundLib(first, tenant.Default) require.NotNil(t, installed) @@ -93,7 +93,7 @@ func TestBind_UnstatableLibraryIsThatTenantsUnavailable(t *testing.T) { require.NoError(t, os.MkdirAll(line, 0o750)) require.NoError(t, os.WriteFile(filepath.Join(line, "manifest.json"), fmt.Appendf(nil, `{"library":"libchtypes.dylib","clickhouse_version":%q,"clickhouse_minor":%q}`, - TestServerVersion+"-lts", testLine), 0o600)) + testServerVersion+"-lts", testLine), 0o600)) link := filepath.Join(line, "libchtypes.dylib") require.NoError(t, os.Symlink(filepath.Join(dir, "gone"), link)) @@ -101,7 +101,7 @@ func TestBind_UnstatableLibraryIsThatTenantsUnavailable(t *testing.T) { require.NoError(t, err) t.Cleanup(eng.Close) - eng.Bind("tokyo", TestServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) + eng.Bind("tokyo", testServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) u := unavailable(t, eng, "tokyo") assert.Empty(t, u.Table) assert.Contains(t, u.Cause, link) @@ -111,12 +111,12 @@ func TestBind_UnstatableLibraryIsThatTenantsUnavailable(t *testing.T) { require.NoError(t, os.Remove(link)) require.NoError(t, os.Symlink(installed.Path, link)) - eng.Bind("tokyo", TestServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) + eng.Bind("tokyo", testServerVersion, "Asia/Tokyo", []*discovery.TableSchema{eventsTable()}) u = unavailable(t, eng, "tokyo") assert.Contains(t, u.Cause, `"UTC"`, "a symlink to the open library is the same image") assert.Contains(t, u.Cause, "one timezone per ClickHouse version line") - eng.Bind(tenant.Default, TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) answers(t, eng, tenant.Default) assert.Same(t, installed, boundLib(eng, tenant.Default)) } @@ -124,7 +124,7 @@ func TestBind_UnstatableLibraryIsThatTenantsUnavailable(t *testing.T) { // TestBind_ZoneRecordIsKeyedOnTheLibrary: the record names the library the // SDK opened, by path, with the zone it was opened in. func TestBind_ZoneRecordIsKeyedOnTheLibrary(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) lib := boundLib(eng, tenant.Default) require.NotNil(t, lib) @@ -139,11 +139,11 @@ func TestBind_ZoneRecordIsKeyedOnTheLibrary(t *testing.T) { // server is logged when the tenant's library changes, not on every refresh, // and a server on another patch of the line is a warning. func TestBind_LogsTheLibraryOncePerTenant(t *testing.T) { - eng := TestEngine(t, eventsTable()) + eng := testEngine(t, eventsTable()) logs := logtest.Capture(t, slog.LevelInfo) - eng.Bind("exact", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) - eng.Bind("exact", TestServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind("exact", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) + eng.Bind("exact", testServerVersion, "UTC", []*discovery.TableSchema{eventsTable()}) otherPatch := testLine + ".1.1" eng.Bind("other", otherPatch, "UTC", []*discovery.TableSchema{eventsTable()}) answers(t, eng, "other") @@ -152,8 +152,8 @@ func TestBind_LogsTheLibraryOncePerTenant(t *testing.T) { exact := withMsg(recs, "chtypes library bound") require.Len(t, exact, 1, "one record for the tenant's first bind, none for the refresh") assert.Equal(t, "exact", exact[0]["tenant"]) - assert.Equal(t, TestServerVersion, exact[0]["server_version"]) - assert.Equal(t, TestServerVersion+"-lts", exact[0]["chtypes_version"]) + assert.Equal(t, testServerVersion, exact[0]["server_version"]) + assert.Equal(t, testServerVersion+"-lts", exact[0]["chtypes_version"]) other := withMsg(recs, "chtypes library bound from another patch of the server's line; verdicts follow the artifact's patch") require.Len(t, other, 1) @@ -171,12 +171,12 @@ var sdkWarnedPatches atomic.Int64 // The Engine itself asks for lines, which never warn; this asks the registry // for an uninstalled patch directly to make the SDK speak. func TestNewEngine_SDKWarningsGoToTheLog(t *testing.T) { - eng := TestEngine(t) + eng := testEngine(t) logs := logtest.Capture(t, slog.LevelInfo) patch := fmt.Sprintf("%s.0.%d", testLine, sdkWarnedPatches.Add(1)) openedZones.mu.Lock() - chtypes.SetDefaultTimezone("UTC") // the zone TestEngine opened the line in + chtypes.SetDefaultTimezone("UTC") // the zone testEngine opened the line in res, err := eng.reg.Resolve(chtypes.Version(patch)) openedZones.mu.Unlock() require.NoError(t, err) From db2152687ca13f5bbf15fc8f5250cb4dc4d699d2 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:50:20 -0400 Subject: [PATCH 32/70] fix(api): cap query-path connections per pool, not per server The structured-query and pipe reader kept one HTTP client per connection cap and TLS config, so plain-HTTP tenants on one server with the same max_open_conns shared a single cap whatever their database or user: one tenant's slow reads could starve another's. Clients are now keyed by the pool a tenant reads through (server URL, user, password, database, TLS config, cap), the tuple the native pools are keyed by, and the cap is the pool's size, so tenants sharing a tuple share one cap sized to the largest ask among them, as the native pool was. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/api/clickhouse_http.go | 50 ++++++++++++------ internal/api/clickhouse_http_test.go | 76 ++++++++++++++++++++++++---- internal/api/pipes.go | 5 +- internal/api/structured_query.go | 5 +- internal/app/wire.go | 21 +++++--- 5 files changed, 118 insertions(+), 39 deletions(-) diff --git a/internal/api/clickhouse_http.go b/internal/api/clickhouse_http.go index 76998330..05da5fbc 100644 --- a/internal/api/clickhouse_http.go +++ b/internal/api/clickhouse_http.go @@ -110,8 +110,8 @@ type chReader struct { build func(tlsCfg *tls.Config, conns int) *http.Client mu sync.Mutex - // clients holds one set of clients per connection cap. - clients map[int]*chconn.HTTPClients + // clients holds one client per pool (readerPool). + clients map[readerPool]*http.Client // maxResponseBytes optionally overrides maxCHResponseBytes. Test-only // seam for the cap-overflow path; not a production knob. @@ -119,7 +119,20 @@ type chReader struct { } func newCHReader(build func(tlsCfg *tls.Config, conns int) *http.Client) *chReader { - return &chReader{build: build, clients: map[int]*chconn.HTTPClients{}} + return &chReader{build: build, clients: map[readerPool]*http.Client{}} +} + +// readerPool is what one client's connections are shared by: the server, the +// credentials and the database — the tuple a native pool is keyed by — the +// TLS config, and the cap. Tenants naming the same tuple share one cap, as +// they share one native pool; a tenant on another database or user on the +// same server gets a cap of its own. The set grows with the pools ever read +// through, not with requests: a client whose pool is gone keeps only its +// transport, whose idle connections time out on their own. +type readerPool struct { + url, username, password, database string + tls *tls.Config + conns int } // sharedCHReader serves both cached read handlers, so a tenant's reads share @@ -128,10 +141,9 @@ func newCHReader(build func(tlsCfg *tls.Config, conns int) *http.Client) *chRead var sharedCHReader = newCHReader(readerHTTPClient) // readerHTTPClient is the read paths' client: net/http's default transport -// with the target's TLS config and at most conns connections per server. -// Tenants on one server with the same cap share it, as tenants on one tuple -// share a pool; a read that finds every connection busy waits for one until -// its deadline. Like the proxy's client it has no Timeout — every request +// with the target's TLS config and at most conns connections to the server, +// one per pool (readerPool); a read that finds every connection busy waits +// for one until its deadline. Like the proxy's client it has no Timeout — every request // carries a deadline — and does not chase redirects: the target is operator // config, and ClickHouse does not redirect in normal operation. func readerHTTPClient(tlsCfg *tls.Config, conns int) *http.Client { @@ -147,19 +159,27 @@ func readerHTTPClient(tlsCfg *tls.Config, conns int) *http.Client { } } -// client returns the client for target under a cap of conns connections. +// client returns the client for target's pool under a cap of conns +// connections. func (c *chReader) client(target chconn.Target, conns int) *http.Client { if conns <= 0 { conns = defaultReadConns } - c.mu.Lock() - clients, ok := c.clients[conns] - if !ok { - clients = chconn.NewHTTPClients(func(tlsCfg *tls.Config) *http.Client { return c.build(tlsCfg, conns) }) - c.clients[conns] = clients + key := readerPool{ + url: target.URL, username: target.Username, password: target.Password, database: target.Database, + tls: target.TLS, conns: conns, } - c.mu.Unlock() - return clients.For(target) + c.mu.Lock() + defer c.mu.Unlock() + if cl, ok := c.clients[key]; ok { + return cl + } + // A copy of the TLS config, never the target's own: net/http appends its + // HTTP/2 protocols to NextProtos in place when a transport first dials, + // which would race the driver's handshakes on the shared config. + cl := c.build(target.TLS.Clone(), conns) + c.clients[key] = cl + return cl } // chResponseTooLargeError is a response past the read paths' buffer cap. The diff --git a/internal/api/clickhouse_http_test.go b/internal/api/clickhouse_http_test.go index dce0ca84..4d4f98b7 100644 --- a/internal/api/clickhouse_http_test.go +++ b/internal/api/clickhouse_http_test.go @@ -391,21 +391,40 @@ func TestCHReader_Errors(t *testing.T) { }) } -// TestCHReader_ClientsPerCap: a tenant's reads share one transport per -// connection cap, capped at that many connections per server; no cap is -// defaultReadConns, and two caps never share a transport. -func TestCHReader_ClientsPerCap(t *testing.T) { +// TestCHReader_ClientsPerPool: reads share a transport — and so a cap — only +// within one pool: the same server, credentials, database, TLS config and +// cap, the tuple a native pool is keyed by. Two tenants on one plain-HTTP +// server that differ in database or user each get their own cap; no cap is +// defaultReadConns. +func TestCHReader_ClientsPerPool(t *testing.T) { t.Parallel() r := newCHReader(readerHTTPClient) - target := chconn.Target{URL: fakeCHURL} maxConns := func(c *http.Client) int { return c.Transport.(*http.Transport).MaxConnsPerHost } + acme := chconn.Target{URL: fakeCHURL, Username: "acme", Password: "pw", Database: "acme"} + + same := acme + assert.Same(t, r.client(acme, 7), r.client(same, 7), "an identical tuple shares the pool's cap") + assert.Equal(t, 7, maxConns(r.client(acme, 7))) + assert.Equal(t, 7, r.client(acme, 7).Transport.(*http.Transport).MaxIdleConnsPerHost) + + otherDB, otherUser, otherPassword := acme, acme, acme + otherDB.Database = "globex" + otherUser.Username = "globex" + otherPassword.Password = "rotated" + otherServer := acme + otherServer.URL = "http://clickhouse-2.test:8123" + withTLS := acme + withTLS.TLS = &tls.Config{ServerName: "clickhouse.test"} + for name, other := range map[string]chconn.Target{ + "database": otherDB, "user": otherUser, "password": otherPassword, "server": otherServer, "tls": withTLS, + } { + assert.NotSame(t, r.client(acme, 7), r.client(other, 7), "another %s is another pool", name) + assert.Equal(t, 7, maxConns(r.client(other, 7)), name) + } + assert.NotSame(t, r.client(acme, 7), r.client(acme, 8), "another cap is another client") - assert.Equal(t, defaultReadConns, maxConns(r.client(target, 0))) - assert.Same(t, r.client(target, 0), r.client(target, defaultReadConns)) - assert.Equal(t, 7, maxConns(r.client(target, 7))) - assert.Same(t, r.client(target, 7), r.client(target, 7)) - assert.NotSame(t, r.client(target, 7), r.client(target, 8)) - assert.Equal(t, 7, r.client(target, 7).Transport.(*http.Transport).MaxIdleConnsPerHost) + assert.Equal(t, defaultReadConns, maxConns(r.client(acme, 0))) + assert.Same(t, r.client(acme, 0), r.client(acme, defaultReadConns)) } // TestCHReader_ConnectionCapHolds: past the cap, a read waits for a @@ -453,6 +472,41 @@ func TestCHReader_ConnectionCapHolds(t *testing.T) { assert.Equal(t, 1, peak) } +// TestCHReader_TenantsOnOneServerKeepTheirOwnCap: one tenant holding every +// connection its cap allows does not hold up another tenant's read on the same +// server — the cross-tenant starvation a per-server cap would cause. +func TestCHReader_TenantsOnOneServerKeepTheirOwnCap(t *testing.T) { + t.Parallel() + entered := make(chan struct{}, 1) + release := make(chan struct{}) + srv := httptest.NewServer(http.HandlerFunc(func(_ http.ResponseWriter, req *http.Request) { + if req.URL.Query().Get("database") == "acme" { + entered <- struct{}{} + <-release + } + })) + t.Cleanup(srv.Close) + // Before srv.Close on a failure too, which waits for acme's request. + var unblock sync.Once + defer unblock.Do(func() { close(release) }) + + r := newCHReader(readerHTTPClient) + var wg sync.WaitGroup + wg.Go(func() { + _, err := r.do(context.Background(), chconn.Target{URL: srv.URL, Database: "acme"}, 1, chRequest{sql: "SELECT 1"}) + assert.NoError(t, err) + }) + <-entered + + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + defer cancel() + _, err := r.do(ctx, chconn.Target{URL: srv.URL, Database: "globex"}, 1, chRequest{sql: "SELECT 1"}) + require.NoError(t, err, "another tenant's read waited on acme's connection") + + unblock.Do(func() { close(release) }) + wg.Wait() +} + // TestCheckRequestSize: a scalar is refused when its percent-encoded form // passes ClickHouse's per-field limit — measured to the byte, 131072 read and // 131073 refused — and all of them together when they pass what the request diff --git a/internal/api/pipes.go b/internal/api/pipes.go index 9e3bc965..34da7fce 100644 --- a/internal/api/pipes.go +++ b/internal/api/pipes.go @@ -32,9 +32,8 @@ type PipesHandler struct { // (chconn.Pools.Target in production); the zero Target is a tenant on no // pool, a 503. Target func(*settings.Store) chconn.Target - // MaxConns caps the tenant's concurrent reads - // ((*settings.Store).ClickHouse().MaxOpenConns in production); nil or - // non-positive is defaultReadConns. + // MaxConns caps the concurrent reads on the tenant's pool (its native + // pool's size in production); nil or non-positive is defaultReadConns. MaxConns func(*settings.Store) int Cache cache.Cache sf singleflight.Group diff --git a/internal/api/structured_query.go b/internal/api/structured_query.go index df59dd26..211b31b8 100644 --- a/internal/api/structured_query.go +++ b/internal/api/structured_query.go @@ -23,9 +23,8 @@ type StructuredQueryHandler struct { // (chconn.Pools.Target in production); the zero Target is a tenant on no // pool, a 503. Target func(*settings.Store) chconn.Target - // MaxConns caps the tenant's concurrent reads - // ((*settings.Store).ClickHouse().MaxOpenConns in production); nil or - // non-positive is defaultReadConns. + // MaxConns caps the concurrent reads on the tenant's pool (its native + // pool's size in production); nil or non-positive is defaultReadConns. MaxConns func(*settings.Store) int Cache cache.Cache Registry RegistrySource diff --git a/internal/app/wire.go b/internal/app/wire.go index b90868ea..f50c3850 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -375,9 +375,16 @@ func (a *App) registryFor(s *settings.Store) *discovery.SchemaRegistry { // per-call setting rather than a property of the pool it shares. func queryTimeout(s *settings.Store) time.Duration { return s.ClickHouse().QueryTimeout } -// maxOpenConns is the tenant's clickhouse.max_open_conns, which also caps the -// HTTP connections its pipes and structured queries hold. -func maxOpenConns(s *settings.Store) int { return s.ClickHouse().MaxOpenConns } +// readConns caps the HTTP connections the pipes and structured queries of the +// tenants on one pool hold between them: the pool's size, the largest +// clickhouse.max_open_conns among them (as the native pool is sized), or the +// tenant's own when it is on no pool. +func (a *App) readConns(s *settings.Store) int { + if m := a.pools.For(s.Tenant()); m != nil { + return m.Sizes().MaxOpenConns + } + return s.ClickHouse().MaxOpenConns +} // wireTypes opens the type layer: the process's one chtypes registry, which // ingest judges every record with and the stream hub evaluates row filters @@ -1039,13 +1046,13 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { streamHandler.Closing = closing // Pipes and structured queries run over the tenant's HTTP target, where - // ClickHouse renders the rows itself, holding at most the tenant's - // max_open_conns connections to it between them. + // ClickHouse renders the rows itself, holding at most its pool's size in + // connections between them. pipesHandler := api.NewPipesHandler(func(s *settings.Store) pipes.Source { return s }, (*settings.Store).Policy, a.chTargetFor, a.cache, queryTimeout) pipesHandler.Tenants = a.tenants - pipesHandler.MaxConns = maxOpenConns + pipesHandler.MaxConns = a.readConns structuredQueryHandler := api.NewStructuredQueryHandler(a.chTargetFor, a.cache, a.registryFor, (*settings.Store).Policy, (*settings.Store).TimestampBucketSeconds, queryTimeout, (*settings.Store).DefaultMaxRows) - structuredQueryHandler.MaxConns = maxOpenConns + structuredQueryHandler.MaxConns = a.readConns schemaHandler := api.NewSchemaHandler(a.registryFor) schemaHandler.Tenants = a.tenants From 3c6d67468b9a220407410efb5f8cb5258c8103db Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:51:32 -0400 Subject: [PATCH 33/70] perf(typelayer): grow a role shape's handle pool under contention Every insert of a column-restricted or check-injecting role ran on its shape's single handle, so that role's ingest was serialized however many requests were in flight. A role shape's pool now grows on contention the way a base table's does, capped at min(GOMAXPROCS, 4) rather than 8: a table can hold 256 shapes, and a quiet shape still holds one handle. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/typelayer/pool.go | 18 +++++++++++----- internal/typelayer/roletable.go | 5 +++-- internal/typelayer/roletable_test.go | 32 +++++++++++++++++++++++----- 3 files changed, 43 insertions(+), 12 deletions(-) diff --git a/internal/typelayer/pool.go b/internal/typelayer/pool.go index 17a9202e..16942c05 100644 --- a/internal/typelayer/pool.go +++ b/internal/typelayer/pool.go @@ -20,21 +20,29 @@ import ( // process-wide serialization gate never beats one thread (0.83-1.0x). A // compiled handle costs ~40 KiB (~96 KiB warm). That cost multiplies by // tables and by tenants, so a pool starts at one handle and grows only when -// every handle it has is busy (see pool.acquire). Role tables keep one handle -// (roleHandles): hot tenants already spread over distinct role handles, and -// 256 shapes x 8 warm handles would be ~200 MiB. An earlier darwin run showed +// every handle it has is busy (see pool.acquire). An earlier darwin run showed // a flat curve, so treat the size as hardware-dependent and re-measure it on // the deployment hardware. const maxPoolSize = 8 -// roleHandles is the handle count of a role-shape table. -const roleHandles = 1 +// maxRolePoolSize caps a role-shape table's pool. Every insert of a +// column-restricted or check-injecting role runs on its shape's handles, so +// one handle would serialize that role's whole ingest; but a table holds up to +// roleCacheSize shapes, and 256 x 8 warm handles would be ~200 MiB where +// 256 x 4 is half that. Like a base table, a shape grows past one handle only +// under contention, so a quiet shape keeps one. +const maxRolePoolSize = 4 // poolSize is how many identical handles one base-table shape may grow to. func poolSize() int { return max(min(runtime.GOMAXPROCS(0), maxPoolSize), 1) } +// rolePoolSize is how many identical handles one role shape may grow to. +func rolePoolSize() int { + return max(min(runtime.GOMAXPROCS(0), maxRolePoolSize), 1) +} + // compileSettings is the fixed parsing profile every table handle is compiled // with. allow_errors_ratio turns "first bad row ends the batch" into // skip-and-continue so every record gets its own verdict; skip_unknown_fields=0 diff --git a/internal/typelayer/roletable.go b/internal/typelayer/roletable.go index d035381b..145de933 100644 --- a/internal/typelayer/roletable.go +++ b/internal/typelayer/roletable.go @@ -19,7 +19,8 @@ import ( // roleCacheSize bounds the per-role shapes held per table. A shape's Defaults // values come from tenant claims and are baked into the compiled handle, so an // unbounded cache is a memory and CPU denial of service — the same reason -// filterCache is bounded. Each entry costs roleHandles (1) compiled handle. +// filterCache is bounded. Each entry holds one compiled handle, and grows to +// rolePoolSize only under contention. const roleCacheSize = 256 // RoleShape is the projection of a table a role may insert through. It is the @@ -172,7 +173,7 @@ func (t *Table) compileRole(shape RoleShape) (*Table, string) { if rerr != nil { return nil, "cannot reconstruct role column declarations: " + rerr.Error() } - p, cause := newPool(t.lib, ddl, roleHandles) + p, cause := newPool(t.lib, ddl, rolePoolSize()) if cause != "" { return nil, cause } diff --git a/internal/typelayer/roletable_test.go b/internal/typelayer/roletable_test.go index e19a3f11..6906b5f3 100644 --- a/internal/typelayer/roletable_test.go +++ b/internal/typelayer/roletable_test.go @@ -2,6 +2,7 @@ package typelayer import ( "encoding/json" + "runtime" "testing" "github.com/stretchr/testify/assert" @@ -284,14 +285,35 @@ func TestRoleTable_BoundedUnderTenantValueChurn(t *testing.T) { assert.Equal(t, `[1, "acme", "", 5]`, string(batch.Rows[0].Line)) } -// TestRoleTable_HasItsOwnHandlePool: a role shape holds its own single handle, -// not the base table's pool (256 shapes x the pool would be unbounded memory). -func TestRoleTable_HasItsOwnHandlePool(t *testing.T) { +// TestRoleTable_PoolGrowsUnderContentionToALowerCap: a role shape has its own +// pool, not the base table's. A quiet shape holds one handle; concurrent +// inserts through it grow the pool like a base table's — one handle would +// serialize the role's whole ingest — but only to min(GOMAXPROCS, 4), since a +// table holds up to roleCacheSize shapes. +func TestRoleTable_PoolGrowsUnderContentionToALowerCap(t *testing.T) { + defer runtime.GOMAXPROCS(runtime.GOMAXPROCS(8)) eng := testEngine(t, ordersTable()) tbl := roleTableFor(t, eng, RoleShape{Defaults: map[string]string{"tenant": "acme"}}) - assert.Len(t, tbl.pool.list(), roleHandles) - assert.Equal(t, int64(roleHandles), tbl.pool.limit.Load(), "a role shape never grows past its one handle") assert.Nil(t, tbl.roles, "a projection is never itself projected") + + p := tbl.pool + require.Len(t, p.list(), 1, "a quiet shape holds one handle") + assert.Equal(t, int64(maxRolePoolSize), p.limit.Load(), "capped below the base table's %d", maxPoolSize) + + held := make([]*schemaSlot, 0, maxRolePoolSize+1) + for range maxRolePoolSize + 1 { + held = append(held, p.acquire()) + } + assert.Len(t, p.list(), maxRolePoolSize, "busy handles grow the pool to its cap and no further") + for _, s := range held { + p.release(s) + } + + // The grown handles answer like the first: the injected default included. + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":1,"amount":5}`+"\n")) + require.NoError(t, err) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, `[1, "acme", "", 5]`, string(batch.Rows[0].Line)) } func (c *roleCache) len() int { From 60862a8a19de8f47604015ff3d42c7faff81f0d3 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:58:49 -0400 Subject: [PATCH 34/70] chore(api): note that the 64 MiB response cap covers every query path The structured query and pipes buffer ClickHouse's answer under the same cap as the raw-SQL proxy; a pipe has no row limit to keep it under, so a large pipe answers 502 clickhouse.response_too_large. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/api/query.go | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/internal/api/query.go b/internal/api/query.go index fa782b0c..0c5af037 100644 --- a/internal/api/query.go +++ b/internal/api/query.go @@ -80,7 +80,10 @@ const ( // genuinely-large results, or the structured query endpoint with its // DefaultMaxRows cap). The cap is here as a safety net against a // runaway SELECT exhausting the API server's RAM; admin-only doesn't - // mean operators won't accidentally OOM themselves. + // mean operators won't accidentally OOM themselves. The structured query + // and pipes buffer under the same cap (chReader), and a pipe has no row + // limit to keep it under: past it, every path answers 502 + // clickhouse.response_too_large. maxCHResponseBytes = 64 << 20 // 64 MiB // maxRequestBodyBytes caps the inbound SQL request body. 16 MiB is well From 84642089b5c25df763669f0357eb937a548e0242 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:59:00 -0400 Subject: [PATCH 35/70] fix(api): render DateTime as RFC 3339 UTC on every read and stream surface /v1/query and pipes pinned date_time_output_format=simple, and the type layer's export inherited it, so a DateTime reached clients as a zone-less `YYYY-MM-DD hh:mm:ss` in the column's or server's zone. A client cannot know that zone, so it could neither order those values against an RFC 3339 instant nor read them as one; the SDK's live-query seam and its client-side stream filters compared them as strings and got it wrong. Both now render ISO: `YYYY-MM-DDThh:mm:ss[.fff]Z`, in UTC whatever the column's zone, at the column's scale. The SSE wire, /v1/query and pipes carry the same bytes, as before, and the worker's best_effort INSERT of the published row stores the exact instant. Date and Date32 are unchanged. The cached-read rendering tag moves to JSONEachRow/2 so no cache serves the old spelling across a rolling deploy. Measured on ClickHouse 26.8.15.10 and the 26.8 chtypes artifact: DateTime, DateTime64(0/3/6/9), explicit-zone columns, Nullable, Array and Tuple render `...Z` in UTC, byte-identical between the server and the artifact (Map, LowCardinality and Dynamic on the server too); an export re-parses to itself; and an INSERT of the ISO rows under best_effort (also under a non-UTC session zone) stores every value unchanged. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/api/clickhouse_http.go | 14 ++++++---- internal/api/clickhouse_http_test.go | 7 +++-- internal/api/ingest_test.go | 19 ++++++------- internal/typelayer/ingest.go | 6 ++++ internal/typelayer/ingest_test.go | 28 +++++++++++++++++++ tests/e2e/sdk/query.test.ts | 17 +++++------ tests/e2e/sdk/streaming.test.ts | 6 ++-- tests/integration/query_types_test.go | 5 ++-- tests/integration/rowfilter_stream_test.go | 2 ++ .../integration/testdata/query_types_pin.json | 2 +- tests/integration/typelayer_wire_test.go | 10 ++++--- 11 files changed, 81 insertions(+), 35 deletions(-) diff --git a/internal/api/clickhouse_http.go b/internal/api/clickhouse_http.go index 05da5fbc..95f8f2a4 100644 --- a/internal/api/clickhouse_http.go +++ b/internal/api/clickhouse_http.go @@ -22,7 +22,7 @@ import ( // every cached read's key (queryCacheKey), so two builds that render rows // differently never serve each other's entries from a shared cache during a // rolling deploy. Change it with any change to the rendering settings below. -const chRendering = "JSONEachRow/1" +const chRendering = "JSONEachRow/2" // chReadSettingsFixed go on every request the cached read paths send, ahead // of the role's caps. @@ -31,9 +31,13 @@ const chRendering = "JSONEachRow/1" // DateTime64's scale, an Enum's name and an IPv6's compression are the // server's own. Every knob that changes the bytes is pinned rather than // inherited, because a tenant's server or profile may set any of them: -// 64-bit integers and decimals as bare numbers, NaN and Inf as null, -// DateTime as `YYYY-MM-DD hh:mm:ss[.fff]` (the spelling the SSE wire carries, -// #372), and a named tuple as an object. +// 64-bit integers and decimals as bare numbers, NaN and Inf as null, a named +// tuple as an object, and DateTime as RFC 3339 in UTC, +// `YYYY-MM-DDThh:mm:ss[.fff]Z` with the column's scale, whatever the column's +// or the server's zone — the spelling the SSE wire carries (typelayer's +// export pins the same), so neither a client nor the cache needs to know the +// server's zone (#372). Date and Date32 are unaffected. Measured on +// 26.8.15.10. // // Failure: wait_end_of_query buffers the result server-side until the query // has finished, and http_write_exception_in_output_format=0 keeps an @@ -57,7 +61,7 @@ var chReadSettingsFixed = map[string]string{ "output_format_json_quote_64bit_integers": "0", "output_format_json_quote_decimals": "0", "output_format_json_quote_denormals": "0", - "date_time_output_format": "simple", + "date_time_output_format": "iso", "output_format_json_named_tuples_as_objects": "1", } diff --git a/internal/api/clickhouse_http_test.go b/internal/api/clickhouse_http_test.go index 4d4f98b7..f7f9a73a 100644 --- a/internal/api/clickhouse_http_test.go +++ b/internal/api/clickhouse_http_test.go @@ -216,8 +216,8 @@ func TestCHReader_ResponseShape(t *testing.T) { // The bytes are ClickHouse's: key order, a decimal's digits and a // timestamp's spelling all pass through untouched. name: "rows are copied, not re-encoded", - body: "{\"z\":1,\"a\":12.50,\"ts\":\"2026-01-15 10:30:00.120\"}\n", - want: `[{"z":1,"a":12.50,"ts":"2026-01-15 10:30:00.120"}]`, + body: "{\"z\":1,\"a\":12.50,\"ts\":\"2026-01-15T10:30:00.120Z\"}\n", + want: `[{"z":1,"a":12.50,"ts":"2026-01-15T10:30:00.120Z"}]`, }, } for _, tt := range tests { @@ -255,6 +255,9 @@ func TestCHReader_Request(t *testing.T) { for name, want := range chReadSettingsFixed { assert.Equal(t, want, got.query.Get(name), name) } + // Pinned here rather than read from the map: the SSE wire (typelayer's + // export) and the SDK compare against this spelling. + assert.Equal(t, "iso", got.query.Get("date_time_output_format"), "DateTime as RFC 3339 in UTC") assert.Equal(t, "2", got.query.Get("readonly")) assert.Equal(t, "warehouse", got.query.Get("database")) assert.Equal(t, []string{"/home", `a\tb`}, ch.params()) diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index 050bb690..1a6ee1c9 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -2590,10 +2590,9 @@ func publishedRow(t *testing.T, payload []byte) map[string]any { // TestIngest_TimestampsCanonicalized is the #372 contract, now satisfied by // construction: whatever spelling a producer uses, the published payload — the // one copy SSE subscribers, the ClickHouse insert and the DLQ all consume — -// carries the instant as ClickHouse's OWN writer renders it. That is -// "2026-06-21 04:00:00" in the column's zone, not RFC 3339 with a Z: the row is -// the server's rendering of the stored value, so the SSE frame and a -// /v1/query row cannot disagree. +// carries the instant as ClickHouse's OWN writer renders it, under the same +// date_time_output_format=iso the query paths pin: RFC 3339 in UTC at the +// column's scale, so the SSE frame and a /v1/query row cannot disagree. func TestIngest_TimestampsCanonicalized(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} @@ -2609,10 +2608,10 @@ func TestIngest_TimestampsCanonicalized(t *testing.T) { require.Equal(t, http.StatusOK, w.Code) data := publishedData(t, pub) - assert.Equal(t, "2026-06-21 04:00:00", data["ts"]) + assert.Equal(t, "2026-06-21T04:00:00Z", data["ts"]) // Sub-second digits are the column's precision, zeros and all — the server // does not trim them the way the old canonicalizer did. - assert.Equal(t, "2026-06-21 04:00:00.500", data["ts_ms"]) + assert.Equal(t, "2026-06-21T04:00:00.500Z", data["ts_ms"]) assert.Equal(t, "e", data["name"], "non-timestamp columns untouched") } @@ -2626,7 +2625,7 @@ func TestIngest_AutoInjectedLiteralTimestampCanonicalized(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} h := newTestIngestHandler(t, tsRegistry(t), pub) - staticTS := "2026-06-21T04:00:00Z" + staticTS := "2026-06-21T06:00:00+02:00" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "events": { @@ -2648,7 +2647,7 @@ func TestIngest_AutoInjectedLiteralTimestampCanonicalized(t *testing.T) { h.Handle(w, withTenant(req)) require.Equal(t, http.StatusOK, w.Code) - assert.Equal(t, "2026-06-21 04:00:00", publishedData(t, pub)["ts"], + assert.Equal(t, "2026-06-21T04:00:00Z", publishedData(t, pub)["ts"], "auto-injected literal must be parsed, not published in its policy spelling") } @@ -2703,8 +2702,8 @@ func TestIngest_Batch_MixedTimestampSpellings(t *testing.T) { spellings = append(spellings, publishedRow(t, msg.Data)["ts"].(string)) } assert.Equal(t, []string{ - "2026-06-21 04:00:00", // RFC 3339 in - "2026-06-21 04:00:00", // Unix seconds in — same instant, same rendering + "2026-06-21T04:00:00Z", // RFC 3339 in + "2026-06-21T04:00:00Z", // Unix seconds in — same instant, same rendering }, spellings) } diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go index 439d751c..7dbdb40b 100644 --- a/internal/typelayer/ingest.go +++ b/internal/typelayer/ingest.go @@ -34,8 +34,14 @@ type IngestOptions struct { // detection off when the caller asked for strict positional CSV/TSV. The worker // inserts JSONCompactEachRow, so the detect_header settings have no real-INSERT // twin to keep in step. ok is false for a format Ingest does not parse. +// +// The export renders DateTime as RFC 3339 in UTC (`…T…Z`, the column's scale), +// the spelling the query paths pin too, so a published row reads the same as a +// queried one and carries its instant whatever the zone; the worker's +// best_effort INSERT stores that exact instant. Measured on 26.8.15.10. func parseSettings(format Format, opts IngestOptions) (settings map[string]string, ok bool) { settings = InsertSettings() + settings["date_time_output_format"] = "iso" switch format { case FormatJSONEachRow, FormatCSVWithNames, FormatTSVWithNames: case FormatCSV: diff --git a/internal/typelayer/ingest_test.go b/internal/typelayer/ingest_test.go index 5b72e213..af576e37 100644 --- a/internal/typelayer/ingest_test.go +++ b/internal/typelayer/ingest_test.go @@ -171,6 +171,34 @@ func TestInsertSettings_BestEffortDateTime(t *testing.T) { assert.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) } +// TestIngest_DateTimeExportsAsRFC3339UTC: the exported row — what is published +// to the stream and inserted by the worker — spells every DateTime as RFC 3339 +// in UTC at the column's scale, whatever the input spelling or the column's +// zone, as the query paths render it; Date and Date32 keep their own form. +func TestIngest_DateTimeExportsAsRFC3339UTC(t *testing.T) { + eng := TestEngine(t, &discovery.TableSchema{Name: "times", Columns: []discovery.Column{ + {Name: "dt", Type: "DateTime", Position: 1}, + {Name: "dt3", Type: "DateTime64(3)", Position: 2}, + {Name: "dt6", Type: "DateTime64(6)", Position: 3}, + {Name: "dtz", Type: "DateTime('Asia/Tokyo')", Position: 4}, + {Name: "dt3z", Type: "DateTime64(3, 'America/New_York')", Position: 5}, + {Name: "d", Type: "Date", Position: 6}, + {Name: "d32", Type: "Date32", Position: 7}, + }}) + tbl, err := eng.Table(tenant.Default, "times") + require.NoError(t, err) + t.Cleanup(tbl.Release) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"dt":"2026-03-24 12:00:00","dt3":"2026-03-24T14:00:00+02:00",`+ + `"dt6":"2026-03-24 12:00:00.123456","dtz":"2026-03-24 21:00:00","dt3z":"2026-03-24 08:00:00.12",`+ + `"d":"2026-03-24","d32":"1960-01-02"}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, `["2026-03-24T12:00:00Z", "2026-03-24T12:00:00.000Z", "2026-03-24T12:00:00.123456Z", `+ + `"2026-03-24T12:00:00Z", "2026-03-24T12:00:00.120Z", "2026-03-24", "1960-01-02"]`, string(batch.Rows[0].Line)) +} + func TestInsertSettings_ReturnsAFreshMap(t *testing.T) { t.Parallel() a := InsertSettings() diff --git a/tests/e2e/sdk/query.test.ts b/tests/e2e/sdk/query.test.ts index fc336cb0..7f237160 100644 --- a/tests/e2e/sdk/query.test.ts +++ b/tests/e2e/sdk/query.test.ts @@ -240,9 +240,9 @@ describe("Query", () => { // The value SPELLING is part of the SDK contract, and none of the suite // tables can show it: they are all String/UInt32/DateTime64. A Decimal // arrives as a JSON number (`wavehouse codegen` types it as one), and a - // DateTime64 keeps ClickHouse's own `YYYY-MM-DD HH:MM:SS.fff` spelling - // rather than ISO-8601 — the same bytes the stream carries (#372). - it("renders a Decimal as a number and a DateTime64 in ClickHouse's spelling", async () => { + // DateTime64 as RFC 3339 in UTC at the column's scale, `.000Z` and all — + // the same bytes the stream carries (#372). + it("renders a Decimal as a number and a DateTime64 as RFC 3339 UTC", async () => { const admin = adminClient(); const t = `types_${testId().replace(/-/g, "_")}`; @@ -267,7 +267,7 @@ describe("Query", () => { const row = result.data![0] as Record; expect(typeof row.amount).toBe("number"); expect(row.amount).toBe(12.5); - expect(row.at).toBe("2026-01-15 10:30:00.123"); + expect(row.at).toBe("2026-01-15T10:30:00.123Z"); } finally { await setPolicy(currentPolicy); await chQuery(`DROP TABLE IF EXISTS default.\`${t}\``); @@ -276,8 +276,8 @@ describe("Query", () => { // ClickHouse parses a filter value on a DateTime column itself, so RFC 3339 // with any offset is an exact instant, a zone-less value reads in the - // column's own zone, and a timestamp read back from /v1/query, in - // ClickHouse's own spelling, filters as it is. + // column's own zone, and a timestamp read back from /v1/query — RFC 3339 in + // UTC, whatever the column's zone — filters as it is. it("filters DateTime and DateTime64 columns by RFC 3339 and by the spelling it returns", async () => { const admin = adminClient(); const t = `ts_${testId().replace(/-/g, "_")}`; @@ -323,8 +323,9 @@ describe("Query", () => { const back = await wh.from(t).select("id", "at", "at3", "atk").where("id", "=", "r1").fetch(); expect(back.error).toBeNull(); const row = back.data![0] as { at: string; at3: string; atk: string }; - expect(row.at).toBe("2026-01-15 10:30:00"); - expect(row.atk).toBe("2026-01-15 19:30:00"); + expect(row.at).toBe("2026-01-15T10:30:00Z"); + expect(row.at3).toBe("2026-01-15T10:30:00.123Z"); + expect(row.atk).toBe("2026-01-15T10:30:00Z"); expect(await ids("at", "=", row.at)).toEqual(["r1"]); expect(await ids("at3", "=", row.at3)).toEqual(["r1"]); expect(await ids("atk", "=", row.atk)).toEqual(["r1"]); diff --git a/tests/e2e/sdk/streaming.test.ts b/tests/e2e/sdk/streaming.test.ts index d0a5f51c..83aa9946 100644 --- a/tests/e2e/sdk/streaming.test.ts +++ b/tests/e2e/sdk/streaming.test.ts @@ -135,9 +135,9 @@ describe("Streaming", () => { await waitForCondition(() => receivedEvents.some((e) => e.data?.event_id === id), 10_000); const frame = receivedEvents.find((e) => e.data?.event_id === id); - // The column's own rendering: "YYYY-MM-DD hh:mm:ss.SSS" in its zone - // (DateTime64(3) here), not RFC 3339 with a Z. - expect(frame?.data.received_timestamp).toBe("2026-06-21 04:00:00.123"); + // RFC 3339 in UTC at the column's scale (DateTime64(3) here), the + // spelling /v1/query renders too. + expect(frame?.data.received_timestamp).toBe("2026-06-21T04:00:00.123Z"); // The ClickHouse insert is async behind the stream event — poll the query // path until the row lands, then compare the two renderings. diff --git a/tests/integration/query_types_test.go b/tests/integration/query_types_test.go index 309fcda3..a7aeda82 100644 --- a/tests/integration/query_types_test.go +++ b/tests/integration/query_types_test.go @@ -68,8 +68,9 @@ const queryTypesRow = `( // change and belongs in the CHANGELOG. // // ClickHouse renders the body (FORMAT JSONEachRow under the reader's pinned -// output settings): keys in SELECT order, Decimal as a JSON number. Measured -// identical on 26.6.3.62 and 26.8.7.19. +// output settings): keys in SELECT order, Decimal as a JSON number, DateTime +// as RFC 3339 in UTC at the column's scale (date_time_output_format=iso). +// Measured on 26.8.15.10. func TestQuery_TypeRendering_Pin(t *testing.T) { e := env(t) diff --git a/tests/integration/rowfilter_stream_test.go b/tests/integration/rowfilter_stream_test.go index c77acc99..e086f3fc 100644 --- a/tests/integration/rowfilter_stream_test.go +++ b/tests/integration/rowfilter_stream_test.go @@ -622,6 +622,8 @@ func rowFilterStoredLine(t *testing.T, table string, id uint32) []byte { q.Set("param_target_table", table) q.Set("param_id", fmt.Sprint(id)) q.Set("query", "SELECT * FROM {target_table:Identifier} WHERE id = {id:UInt32} FORMAT JSONCompactEachRow") + // The DateTime spelling a published row carries (typelayer's export). + q.Set("date_time_output_format", "iso") body := rowFilterCH(t, http.MethodGet, q, nil) line := strings.TrimRight(string(body), "\n") require.NotEmpty(t, line, "stored row must come back") diff --git a/tests/integration/testdata/query_types_pin.json b/tests/integration/testdata/query_types_pin.json index 58f2f03f..ea3ee9f1 100644 --- a/tests/integration/testdata/query_types_pin.json +++ b/tests/integration/testdata/query_types_pin.json @@ -1 +1 @@ -[{"i8":-8,"i16":-16,"i32":-32,"i64":-9007199254740993,"u8":8,"u16":16,"u32":32,"u64":18446744073709551615,"f32":0.1,"f64":0.1,"dec":12.5,"dec64":1.5,"s":"hello","ls":"lc","fs":"ab\u0000\u0000","uu":"11111111-2222-3333-4444-555555555555","en":"a","bl":true,"ip4":"10.0.0.1","ip6":"::1","d":"2026-01-15","dt":"2026-01-15 10:30:00","dt64":"2026-01-15 10:30:00.123","arr":["a","b"],"m":{"k":7},"nn":42,"nnull":null}] +[{"i8":-8,"i16":-16,"i32":-32,"i64":-9007199254740993,"u8":8,"u16":16,"u32":32,"u64":18446744073709551615,"f32":0.1,"f64":0.1,"dec":12.5,"dec64":1.5,"s":"hello","ls":"lc","fs":"ab\u0000\u0000","uu":"11111111-2222-3333-4444-555555555555","en":"a","bl":true,"ip4":"10.0.0.1","ip6":"::1","d":"2026-01-15","dt":"2026-01-15T10:30:00Z","dt64":"2026-01-15T10:30:00.123Z","arr":["a","b"],"m":{"k":7},"nn":42,"nnull":null}] diff --git a/tests/integration/typelayer_wire_test.go b/tests/integration/typelayer_wire_test.go index 77bd6ff9..b14e2d8a 100644 --- a/tests/integration/typelayer_wire_test.go +++ b/tests/integration/typelayer_wire_test.go @@ -91,8 +91,8 @@ func TestTypelayerWire_PublishedRowIsTheStoredRow(t *testing.T) { // And spot-check that these really are the coerced values, not the // producer's spellings passed through. require.Len(t, wire, 7) - assert.Equal(t, "2026-06-21 04:00:00", wire[1], "offset applied, rendered in the column's zone") - assert.Equal(t, "2026-06-21 04:00:00.123", wire[2], "ticks at the column's precision") + assert.Equal(t, "2026-06-21T04:00:00Z", wire[1], "offset applied, rendered as RFC 3339 in UTC") + assert.Equal(t, "2026-06-21T04:00:00.123Z", wire[2], "ticks at the column's precision") assert.EqualValues(t, 0, wire[4], "256 into a UInt8 wraps — the stored truth") } @@ -148,13 +148,15 @@ func firstDataRow(t *testing.T, body interface{ Read([]byte) (int, error) }) ([] } // selectJSONCompactRow reads one row back through ClickHouse's HTTP interface in -// JSONCompactEachRow — the same writer that produced the published row, so the -// two renderings are comparable without a second interpretation step. +// JSONCompactEachRow under the DateTime spelling the export pins — the same +// writer and settings that produced the published row, so the two renderings +// are comparable without a second interpretation step. func selectJSONCompactRow(t *testing.T, chHTTPURL, query string) []any { t.Helper() q := url.Values{} q.Set("database", testCHDatabase) q.Set("query", query+" FORMAT JSONCompactEachRow") + q.Set("date_time_output_format", "iso") req, err := http.NewRequestWithContext(context.Background(), http.MethodGet, chHTTPURL+"?"+q.Encode(), nil) require.NoError(t, err) From a8886612eb5852ff0098e4460bdeaad113e5e602 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 07:59:27 -0400 Subject: [PATCH 36/70] fix(sdk): compare timestamps as instants in stream filters and liveQuery Client-side stream filters and the liveQuery backfill seam compared timestamps as strings, which is right only for one zone at one scale. The server renders a DateTime64(3) as `...00.000Z` while a filter may say `...00Z` or use an offset, and an SSE envelope's timestamp carries up to nine trimmed fraction digits: `...00Z` sorts after `...00.123Z`. Two strings that both read as timestamps are now compared as instants, to the nanosecond, for eq, neq, in and the ordered operators; a value with no zone reads as UTC. The backfill seam compares at the coarser of the two precisions, so an event in the boundary row's own millisecond is covered by it. Anything else keeps strict equality and string order. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- clients/ts/src/query-builder.test.ts | 40 +++++++++++ clients/ts/src/query-builder.ts | 33 +++++++-- clients/ts/src/stream/live-query.test.ts | 43 ++++++++++++ clients/ts/src/stream/live-query.ts | 15 +++- clients/ts/src/timestamp.test.ts | 59 ++++++++++++++++ clients/ts/src/timestamp.ts | 88 ++++++++++++++++++++++++ 6 files changed, 271 insertions(+), 7 deletions(-) create mode 100644 clients/ts/src/timestamp.test.ts create mode 100644 clients/ts/src/timestamp.ts diff --git a/clients/ts/src/query-builder.test.ts b/clients/ts/src/query-builder.test.ts index 63dfd728..7f34543b 100644 --- a/clients/ts/src/query-builder.test.ts +++ b/clients/ts/src/query-builder.test.ts @@ -396,6 +396,46 @@ describe("QueryBuilder", () => { expect(received[0].page).toBe("/home"); }); + it("filters a stream's timestamps by instant, not by spelling", () => { + // The server renders a DateTime64(3) as `…00.000Z`; a filter may say `…00Z` + // or use an offset. Compared as strings, `…00.000Z` < `…00Z`. + let next: ((e: any) => void) | undefined; + mockCreateStream.mockReturnValue({ + subscribe(sub: any) { + next = sub.next; + return () => {}; + }, + close() {}, + } as any); + const seen = (b: QueryBuilder, ts: string[]): string[] => { + const got: string[] = []; + b.stream().subscribe({ next: (e: any) => got.push(e.data.ts) }); + for (const t of ts) next?.({ table: "clicks", timestamp: t, data: { ts: t } }); + return got; + }; + const rows = [ + "2026-01-15T10:29:59.999Z", + "2026-01-15T10:30:00.000Z", + "2026-01-15T10:30:00.001Z", + ]; + + expect(seen(builder().where("ts", ">", "2026-01-15T10:30:00Z"), rows)).toEqual([rows[2]]); + expect(seen(builder().where("ts", ">=", "2026-01-15T12:30:00+02:00"), rows)).toEqual( + rows.slice(1), + ); + expect(seen(builder().where("ts", "<", "2026-01-15T10:30:00Z"), rows)).toEqual([rows[0]]); + expect(seen(builder().where("ts", "=", "2026-01-15T10:30:00Z"), rows)).toEqual([rows[1]]); + expect(seen(builder().where("ts", "!=", "2026-01-15T10:30:00Z"), rows)).toEqual([ + rows[0], + rows[2], + ]); + expect(seen(builder().where("ts", "in", ["2026-01-15T10:30:00.001+00:00"]), rows)).toEqual([ + rows[2], + ]); + // Strings that are not timestamps keep strict equality and string order. + expect(seen(builder().where("ts", ">", "/a"), ["/b", "/a"])).toEqual(["/b"]); + }); + // --- Complex chain --- it("builds a complex query", async () => { diff --git a/clients/ts/src/query-builder.ts b/clients/ts/src/query-builder.ts index e42e7031..a2e064b6 100644 --- a/clients/ts/src/query-builder.ts +++ b/clients/ts/src/query-builder.ts @@ -3,6 +3,7 @@ import { request } from "./http.js"; import type { StreamTransport } from "./stream/controller.js"; import { StreamController } from "./stream/controller.js"; import { LiveQuery } from "./stream/live-query.js"; +import { compareInstants } from "./timestamp.js"; import type { Aggregation, FilterOp, @@ -325,9 +326,9 @@ function matchesFilters(row: Record, filters: QueryFilter[]): b function evaluateFilter(actual: unknown, op: string, expected: unknown): boolean { switch (op) { case "eq": - return actual === expected; + return sameValue(actual, expected); case "neq": - return actual !== expected; + return !sameValue(actual, expected); case "gt": return compareOrdered(actual, expected, (a, b) => a > b); case "gte": @@ -337,7 +338,7 @@ function evaluateFilter(actual: unknown, op: string, expected: unknown): boolean case "lte": return compareOrdered(actual, expected, (a, b) => a <= b); case "in": - return Array.isArray(expected) && expected.includes(actual); + return Array.isArray(expected) && expected.some((v) => sameValue(actual, v)); case "like": { if (typeof actual !== "string" || typeof expected !== "string") return false; // Convert SQL LIKE pattern to regex: % → .*, _ → . @@ -356,11 +357,29 @@ function evaluateFilter(actual: unknown, op: string, expected: unknown): boolean } } +/** + * @internal Equality for `eq`, `neq` and `in`: strict, except that two strings + * that both read as timestamps are equal when they name the same instant — a + * `DateTime64(3)` value renders as `…00.000Z`, which a filter written `…00Z` + * must still match, as it does on the server. + */ +function sameValue(actual: unknown, expected: unknown): boolean { + if (actual === expected) return true; + return ( + typeof actual === "string" && + typeof expected === "string" && + compareInstants(actual, expected) === 0 + ); +} + /** * @internal Apply an ordered comparison only when both sides are the same - * comparable primitive (number-vs-number or string-vs-string — strings are - * lexicographic, which is correct for ISO-8601 timestamps). Mismatched or - * unsupported types return false instead of relying on JS coercion. + * comparable primitive (number-vs-number or string-vs-string). Two strings + * that both read as timestamps are ordered by instant (`compareInstants`): + * lexicographic order is right only for the same zone at the same scale, and + * the server renders `…00.000Z` where a filter may say `…00Z` or use an + * offset. Other strings are lexicographic. Mismatched or unsupported types + * return false instead of relying on JS coercion. */ function compareOrdered( actual: unknown, @@ -371,6 +390,8 @@ function compareOrdered( return cmp(actual, expected); } if (typeof actual === "string" && typeof expected === "string") { + const order = compareInstants(actual, expected); + if (order !== undefined) return cmp(order, 0); // `>`/`<` on strings is lexicographic; reuse the same comparator by // casting through `as unknown as number` — the runtime operator works // identically on strings. diff --git a/clients/ts/src/stream/live-query.test.ts b/clients/ts/src/stream/live-query.test.ts index 243186e4..75c8d443 100644 --- a/clients/ts/src/stream/live-query.test.ts +++ b/clients/ts/src/stream/live-query.test.ts @@ -1,6 +1,8 @@ import { afterEach, describe, expect, it, vi } from "vitest"; import { createClient } from "../client.js"; import type { FetchLike, Result, StreamEvent, StreamStatus } from "../types.js"; +import type { StreamController } from "./controller.js"; +import { LiveQuery } from "./live-query.js"; /** * `liveQuery()` opens the stream and runs the REST backfill in the same tick, @@ -118,3 +120,44 @@ describe("liveQuery auth ordering", () => { lq.close(); }); }); + +describe("liveQuery backfill seam", () => { + it("drops buffered events the last row covers, comparing instants", async () => { + // The envelope's timestamp carries up to nine trimmed fraction digits; the + // row's DateTime64(3) column renders `….123Z`. As strings, `…00Z` sorts + // after `…00.123Z` and `….123456789Z` before it; as instants at the + // coarser precision, both are covered and only later events pass. + let push: ((e: StreamEvent) => void) | undefined; + const stream = { + subscribe(sub: { next: (e: StreamEvent) => void }) { + push = sub.next; + return () => {}; + }, + close() {}, + } as unknown as StreamController; + let resolve: ((r: Result[]>) => void) | undefined; + const backfill = new Promise[]>>((r) => { + resolve = r; + }); + const delivered: string[] = []; + new LiveQuery(stream, () => backfill, { next: (e) => delivered.push(e.timestamp) }, []); + + for (const timestamp of [ + "2026-03-24T12:00:00Z", + "2026-03-24T12:00:00.123456789Z", + "2026-03-24T12:00:00.124Z", + "2026-03-24T12:00:01Z", + ]) { + push?.({ table: "clicks", timestamp, data: {} }); + } + resolve?.({ + ok: true, + data: [{ received_timestamp: "2026-03-24T12:00:00.123Z" }], + error: null, + }); + await backfill; + await Promise.resolve(); + + expect(delivered).toEqual(["2026-03-24T12:00:00.124Z", "2026-03-24T12:00:01Z"]); + }); +}); diff --git a/clients/ts/src/stream/live-query.ts b/clients/ts/src/stream/live-query.ts index d9eb6794..7fbc2e47 100644 --- a/clients/ts/src/stream/live-query.ts +++ b/clients/ts/src/stream/live-query.ts @@ -1,3 +1,4 @@ +import { compareInstants } from "../timestamp.js"; import type { QueryFilter, Result, StreamEvent, StreamSubscriber } from "../types.js"; import type { StreamController } from "./controller.js"; @@ -78,7 +79,7 @@ export class LiveQuery> { this._buffering = false; for (const event of this._buffer) { if (this._closed) break; - if (lastTimestamp && event.timestamp <= lastTimestamp) { + if (lastTimestamp && !isAfter(event.timestamp, lastTimestamp)) { continue; // already covered by the fetch } this._subscriber.next(event); @@ -100,3 +101,15 @@ export class LiveQuery> { this._stream.close(); } } + +/** + * Whether an event's timestamp is past the backfill boundary. Compared as + * instants at the coarser of the two precisions: the envelope carries up to + * nanoseconds, a `DateTime64(3)` row milliseconds, so an event in the row's + * own millisecond is covered by it. Two values that are not both timestamps + * fall back to string order. + */ +function isAfter(timestamp: string, boundary: string): boolean { + const order = compareInstants(timestamp, boundary, "coarsest"); + return order === undefined ? timestamp > boundary : order > 0; +} diff --git a/clients/ts/src/timestamp.test.ts b/clients/ts/src/timestamp.test.ts new file mode 100644 index 00000000..afac97e8 --- /dev/null +++ b/clients/ts/src/timestamp.test.ts @@ -0,0 +1,59 @@ +import { describe, expect, it } from "vitest"; +import { compareInstants } from "./timestamp.js"; + +describe("compareInstants", () => { + it("orders the server's spellings by instant, whatever the scale", () => { + // Lexicographically `…00Z` > `…00.000Z` ('Z' > '.'): the same instant. + expect(compareInstants("2026-01-15T10:30:00Z", "2026-01-15T10:30:00.000Z")).toBe(0); + expect(compareInstants("2026-01-15T10:30:00.120Z", "2026-01-15T10:30:00.12Z")).toBe(0); + expect(compareInstants("2026-01-15T10:30:00Z", "2026-01-15T10:30:00.001Z")).toBe(-1); + expect(compareInstants("2026-01-15T10:30:00.5Z", "2026-01-15T10:30:00.499999999Z")).toBe(1); + }); + + it("applies an offset", () => { + expect(compareInstants("2026-01-15T12:30:00+02:00", "2026-01-15T10:30:00.000Z")).toBe(0); + expect(compareInstants("2026-01-15T05:30:00-0500", "2026-01-15T10:30:00Z")).toBe(0); + expect(compareInstants("2026-01-15T19:30:00+09", "2026-01-15T10:30:00Z")).toBe(0); + // 11:00 in +02:00 is 09:00Z — earlier, though its digits sort later. + expect(compareInstants("2026-01-15T11:00:00+02:00", "2026-01-15T10:30:00Z")).toBe(-1); + }); + + it("reads a value with no zone as UTC, with either separator", () => { + expect(compareInstants("2026-01-15 10:30:00", "2026-01-15T10:30:00Z")).toBe(0); + expect(compareInstants("2026-01-15 10:30", "2026-01-15T10:30:00.000Z")).toBe(0); + expect(compareInstants("2026-01-15", "2026-01-15T00:00:00Z")).toBe(0); + }); + + it("compares to the nanosecond", () => { + expect( + compareInstants("2026-01-15T10:30:00.123456789Z", "2026-01-15T10:30:00.123456788Z"), + ).toBe(1); + }); + + it("orders instants before the epoch", () => { + expect(compareInstants("1965-06-01T01:02:03.004Z", "1970-01-01T00:00:00Z")).toBe(-1); + expect(compareInstants("1965-06-01T03:02:03.004+02:00", "1965-06-01T01:02:03.004Z")).toBe(0); + }); + + it("compares at the coarser precision when asked", () => { + // An envelope's nanoseconds against a DateTime64(3) row: the same millisecond. + const row = "2026-03-24T12:00:00.123Z"; + expect(compareInstants("2026-03-24T12:00:00.123456789Z", row, "coarsest")).toBe(0); + expect(compareInstants("2026-03-24T12:00:00.123456789Z", row)).toBe(1); + expect(compareInstants("2026-03-24T12:00:00.124Z", row, "coarsest")).toBe(1); + // A trimmed fraction compares at its own digits, which are exact. + expect(compareInstants("2026-03-24T12:00:00.12Z", row, "coarsest")).toBe(0); + expect(compareInstants("2026-03-24T12:00:00.13Z", row, "coarsest")).toBe(1); + }); + + it("declines anything that is not a timestamp on both sides", () => { + expect(compareInstants("/home", "2026-01-15T10:30:00Z")).toBeUndefined(); + expect(compareInstants("2026-01-15T10:30:00Z", "1768473000")).toBeUndefined(); + expect(compareInstants("2026", "2027")).toBeUndefined(); + expect(compareInstants("2026-13-01", "2026-01-01")).toBeUndefined(); + expect(compareInstants("2026-01-15T24:00:00Z", "2026-01-15T00:00:00Z")).toBeUndefined(); + expect( + compareInstants("2026-01-15T10:30:00.1234567890Z", "2026-01-15T10:30:00Z"), + ).toBeUndefined(); + }); +}); diff --git a/clients/ts/src/timestamp.ts b/clients/ts/src/timestamp.ts new file mode 100644 index 00000000..e13a68c9 --- /dev/null +++ b/clients/ts/src/timestamp.ts @@ -0,0 +1,88 @@ +/** + * Timestamp comparison for the SDK's client-side work: stream `where` filters + * and the live-query backfill seam. + * + * The server renders every DateTime as RFC 3339 in UTC at the column's scale + * (`2026-01-15T10:30:00Z`, `2026-01-15T10:30:00.120Z`), and an SSE envelope's + * `timestamp` carries up to nine trimmed fraction digits. Comparing those as + * strings is right only when both sides share a zone and a scale: `…:00Z` + * sorts after `…:00.000Z` (`Z` > `.`), and a filter value written with an + * offset compares by its digits, not its instant. So two strings that both + * read as timestamps are compared as instants, to the nanosecond. + * + * A value with no zone is read as UTC. The server reads one in the column's + * zone, which the SDK cannot know; UTC is the zone every rendered value is in. + * + * @internal + */ + +/** An instant: whole seconds since the epoch, and nanoseconds into that second. */ +interface Instant { + seconds: number; + nanos: number; + /** The fraction digits as written, which `"coarsest"` truncates to. */ + digits: string; +} + +// YYYY-MM-DD, then optionally a time (`T`, `t` or a space; seconds and a +// fraction of up to nine digits optional), then optionally a zone. +const TIMESTAMP = + /^(\d{4})-(\d{2})-(\d{2})(?:[Tt ](\d{2}):(\d{2})(?::(\d{2})(?:\.(\d{1,9}))?)?)?(Z|z|[+-]\d{2}(?::?\d{2})?)?$/; + +function parseInstant(s: string): Instant | undefined { + const m = TIMESTAMP.exec(s); + if (!m) return undefined; + const [, y, mo, d, h = "0", mi = "0", sec = "0", digits = "", zone = "Z"] = m; + const month = Number(mo); + const day = Number(d); + const hour = Number(h); + const minute = Number(mi); + const second = Number(sec); + if (month < 1 || month > 12 || day < 1 || day > 31 || hour > 23 || minute > 59 || second > 59) { + return undefined; + } + let offsetMinutes = 0; + if (zone !== "Z" && zone !== "z") { + const sign = zone[0] === "-" ? -1 : 1; + const hhmm = zone.slice(1).replace(":", ""); + offsetMinutes = sign * (Number(hhmm.slice(0, 2)) * 60 + Number(hhmm.slice(2) || "0")); + } + // setUTCFullYear rather than Date.UTC, which maps years 0–99 onto 1900–1999. + const date = new Date(0); + date.setUTCFullYear(Number(y), month - 1, day); + date.setUTCHours(hour, minute, second, 0); + const ms = date.getTime(); + if (Number.isNaN(ms)) return undefined; + return { seconds: ms / 1000 - offsetMinutes * 60, nanos: Number(digits.padEnd(9, "0")), digits }; +} + +/** + * Order two timestamps by instant: negative when `a` is earlier, zero when they + * are the same instant, positive when `a` is later, and `undefined` when either + * does not read as a timestamp (the caller falls back to its own comparison). + * + * `"coarsest"` compares at the precision of whichever side writes fewer + * fraction digits, truncating the other: an event stamped `…00.123456789Z` + * against a `DateTime64(3)` row's `…00.123Z` is the same millisecond. `"exact"` + * (the default) compares to the nanosecond, as the server does. + * + * @internal + */ +export function compareInstants( + a: string, + b: string, + precision: "exact" | "coarsest" = "exact", +): number | undefined { + const x = parseInstant(a); + const y = parseInstant(b); + if (!x || !y) return undefined; + if (x.seconds !== y.seconds) return x.seconds < y.seconds ? -1 : 1; + let xn = x.nanos; + let yn = y.nanos; + if (precision === "coarsest") { + const digits = Math.min(x.digits.length, y.digits.length); + xn = Number(x.digits.slice(0, digits).padEnd(9, "0")); + yn = Number(y.digits.slice(0, digits).padEnd(9, "0")); + } + return Math.sign(xn - yn); +} From 4aa01e32d2b4fce60debd72a6d720c05d2b8817f Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 08:07:36 -0400 Subject: [PATCH 37/70] fix(ingest): role shapes keep denied columns, stamp _eq columns, take EPHEMERAL input Three ways the per-role schema disagreed with what the server does: - An `_eq` check on a column the role may not otherwise write refused every insert: the column was dropped from the role's schema, so the injected default could not be expressed. The check column now joins the role's writable set, so an absent value takes the required one and is published with it, a supplied value passes only if it equals it, and any other value fails the check (403). - Denying a column that a DEFAULT, MATERIALIZED or ALIAS expression reads left the role's schema uncompilable and every insert a 503 forever. A denied column is now re-declared MATERIALIZED with what the server stores when an INSERT omits it (its own DEFAULT expression, or defaultValueOfTypeName of its type): naming it is still 117, it stays off the wire, and expressions over it compile and see the stored value. A role shape that still does not compile is a 500 with retryable:false and no Retry-After, since it is a standing property of the policy and the schema that no retry changes; a shape that leaves the role no column to write is refused the same way. - A record naming an EPHEMERAL column was refused (117), where it used to feed the DEFAULT columns over it. JSONEachRow and the *WithNames formats now parse with an explicit column list, the wire columns plus each EPHEMERAL column a DEFAULT reads, so the value feeds those DEFAULTs and the exported row is still the wire columns. One a MATERIALIZED or ALIAS expression reads stays refused: the server computes those itself from the inserted row, which never carries the ephemeral value. One no DEFAULT reads stays refused too: listing it makes the 26.8 artifact reject every record that omits it. Positional CSV/TSV stay the wire columns. A check on an EPHEMERAL column is still refused with 403. The stream's connect-time schema frame now announces the type layer's wire columns, the list envelopes carry, so a table with an EPHEMERAL column no longer gets a drift re-announcement on its first event; the discovery-side "insertable" list that disagreed with it is gone. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/api/ingest.go | 125 +++++++++----- internal/api/ingest_role_test.go | 238 +++++++++++++++++++++++++++ internal/api/ingest_test.go | 28 ++-- internal/discovery/discovery.go | 72 ++------ internal/discovery/discovery_test.go | 68 +------- internal/stream/hub.go | 14 +- internal/stream/hub_test.go | 28 ++-- internal/typelayer/errors.go | 19 +++ internal/typelayer/ingest.go | 23 ++- internal/typelayer/ingest_test.go | 153 ++++++++++++++++- internal/typelayer/roletable.go | 143 +++++++++------- internal/typelayer/roletable_test.go | 136 +++++++++++++-- internal/typelayer/typelayer.go | 171 +++++++++++++++---- 13 files changed, 904 insertions(+), 314 deletions(-) create mode 100644 internal/api/ingest_role_test.go diff --git a/internal/api/ingest.go b/internal/api/ingest.go index 28d6277f..df97f0a9 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -9,7 +9,6 @@ import ( "maps" "math" "net/http" - "reflect" "slices" "sort" "strconv" @@ -169,7 +168,7 @@ type recordReject struct { // store that cannot answer (503) or fails (500), an id another request holds // (503), a type layer that cannot judge the tenant's table (503). // -// Two are not, and retrying either unchanged cannot help: +// Three are not, and retrying any of them unchanged cannot help: // - An insert grant that resolved for the other operation is a 403 and a // caller/config bug. It aborts rather than rejecting per record because the // grant is resolved ONCE per request, so it is true for every record or @@ -179,12 +178,17 @@ type recordReject struct { // or the role lacks, or a name given twice — is a 400 with the // clickhouse.rejected class and ClickHouse's own code (117). The header is // not a record, and no record was read past it. +// - A role whose projection of the table does not compile is a 500 marked +// not retryable (see roleRefusedAbort). type requestAbort struct { Status int Message string Code string // the error class, when one applies (codeCHRejected) ExceptionCode int // ClickHouse's code, when its parser refused the body as a whole RetryAfter string // non-empty → emit a Retry-After header + // Retryable, when set, overrides what a client reads off the status: the + // SDK retries every 5xx unless the body says otherwise. + Retryable *bool } // ingestRun is one request after its body has been ruled on: chtypes' verdict @@ -566,17 +570,23 @@ func (r *batchResult) add(rec *pendingRecord) { // table ClickHouse itself will enforce, plus the predicates the check clauses // become. // -// Columns is the allow/deny decision, answered by omitting the denied columns -// from the compiled schema: a record naming one is then refused per row with -// ClickHouse's own code 117 rather than by a Go walk over the record's keys. -// nil means the role may write every column, which compiles to no second -// handle at all. +// Columns is the allow/deny decision, answered by the compiled schema: a +// column the role may not write is one no INSERT may name, so a record naming +// it is refused per row with ClickHouse's own code 117 rather than by a Go walk +// over the record's keys. nil means the role may write every column, which +// compiles to no second handle at all. // // Defaults is the `_eq` auto-inject: the required value becomes the column's // DEFAULT, so a record that omits it is filled and a record that supplies one // still wins, and is then tested by the filter. An `_in` check has no single // value to inject, so the column keeps the TABLE's own default and the filter // tests that. +// +// An `_eq` column joins Columns even when the role may not otherwise write +// it: the check is what makes a value there legitimate — an absent one takes +// the claim, a supplied one passes only if it equals the claim — and the +// published row must carry it, or the server would store the column's own +// default instead of the value the policy requires. func (h *IngestHandler) insertShape( ctx context.Context, table, role string, @@ -647,6 +657,9 @@ func (h *IngestHandler) insertShape( shape.Defaults = make(map[string]string, len(cols)) } shape.Defaults[col] = s + if shape.Columns != nil && !slices.Contains(shape.Columns, col) { + shape.Columns = append(shape.Columns, col) + } preds = append(preds, policy.Predicate{Column: col, Op: "=", Values: []string{s}}) } } @@ -658,10 +671,10 @@ func (h *IngestHandler) insertShape( // table's own compiled handle instead of a second one. // // Every column is asked through IsColumnAllowed so the allow/deny precedence -// stays in the one place that owns it; the computed kinds are included for the -// same reason, and typelayer keeps them whatever this list says (they are the -// server's to compute, and a MATERIALIZED expression over a dropped column would -// not compile at all). +// stays in the one place that owns it, the computed kinds included. typelayer +// declares every column whatever this list says — a denied one MATERIALIZED, +// so no record may name it and every expression over it still compiles — and +// reads the list for which columns a record may supply. func allowedInsertColumns(schema *discovery.TableSchema, perms *policy.ResolvedPermissions) []string { allowed := make([]string, 0, len(schema.Columns)) for _, c := range schema.Columns { @@ -678,34 +691,25 @@ func allowedInsertColumns(schema *discovery.TableSchema, perms *policy.ResolvedP // scalarString renders a check clause's required value as the string the filter // binds. Every filter parameter binds as {pN:String} whatever the column's // declared type, so this is the only conversion the check path needs. -// -// The reflect.Kind test rather than a type switch is deliberate: policy marks a -// placeholder-free check value with its own string-kinded named type, which -// existed only for a Go-side numeric re-reading ClickHouse now answers. Naming -// the type here would keep it alive; asking for its kind works across its -// removal. func scalarString(v any) (string, bool) { - if s, ok := v.(string); ok { - return s, true - } - rv := reflect.ValueOf(v) - if rv.Kind() == reflect.String { - return rv.String(), true - } - return "", false + s, ok := v.(string) + return s, ok } -// roleTable resolves the compiled handle for this role's projection, or the 503 -// every record of this request gets instead. The type layer being down is an -// outage, never a verdict about the data: a caller must be able to retry the -// same body unchanged. +// roleTable resolves the compiled handle for this role's projection, or the +// abort every record of this request gets instead. The type layer being down +// is an outage, never a verdict about the data: a caller must be able to retry +// the same body unchanged (503). A projection that does not compile is a +// standing condition of the role's policy and the table's schema, not an +// outage (roleRefusedAbort). // // An injected literal the column cannot read (`count UInt64 DEFAULT 'abc'`) is a -// compile refusal, ClickHouse code 6 — measured. That must not become a 503 for -// a policy that is simply unsatisfiable, so the shape is retried without its -// defaults: the check filter then judges the record as sent, which fails closed +// compile refusal, ClickHouse code 6 — measured. That must not refuse every +// insert for a policy that is simply unsatisfiable, so a refused shape is +// retried without its defaults: if that compiles, the defaults were the +// problem, and the check filter judges each record as sent, which fails closed // (an absent column takes the table default and the filter refuses it). The -// type layer logs the refusal once per generation and shape; this adds one +// type layer logs each refusal once per generation and shape; this adds one // rate-limited line saying what was done about it. func (h *IngestHandler) roleTable(ctx context.Context, id tenant.ID, table string, shape typelayer.RoleShape) (*typelayer.Table, *requestAbort) { if h.Types == nil { @@ -716,19 +720,46 @@ func (h *IngestHandler) roleTable(ctx context.Context, id tenant.ID, table strin if err == nil { return tbl, nil } - if len(shape.Defaults) > 0 { + if _, refused := errors.AsType[*typelayer.RoleRefused](err); refused && len(shape.Defaults) > 0 { bare := typelayer.RoleShape{Columns: shape.Columns} - if t, bareErr := h.Types.RoleTable(id, table, bare); bareErr == nil { + t, bareErr := h.Types.RoleTable(id, table, bare) + if bareErr == nil { if h.noticeDue("inject:" + id.String() + "/" + table) { - slog.WarnContext(ctx, "insert check value cannot be injected as a column default; records omitting it will fail the check", + slog.WarnContext(ctx, "the role's schema does not compile with its insert check values as column defaults; "+ + "serving it without them, so records omitting those columns fail the check", "tenant", id, "table", table, "columns", slices.Sorted(maps.Keys(shape.Defaults)), "cause", err.Error()) } return t, nil } + err = bareErr // the refusal that stands without the defaults + } + if refused, ok := errors.AsType[*typelayer.RoleRefused](err); ok { + if h.noticeDue("refused:" + id.String() + "/" + table) { + slog.ErrorContext(ctx, "ingest refused: the role's insert permissions do not compile against this table", + "tenant", id, "table", table, "cause", refused.Cause) + } + return nil, roleRefusedAbort() } return nil, h.typesFailed(ctx, id, table, err) } +// roleRefusedAbort is the answer for a role whose projection of the table does +// not compile. It is a 500 marked not retryable rather than the 503 an outage +// gets, and carries no Retry-After: the refusal follows from the role's policy +// and the table's schema and is cached for the schema generation, so the same +// request fails the same way until an operator changes one of them, and a +// retry hint would only invite a client to hammer it. The body stays generic +// like the 503's — the cause names columns and ClickHouse internals, and is +// the operator's to read in the log. +func roleRefusedAbort() *requestAbort { + retryable := false + return &requestAbort{ + Status: http.StatusInternalServerError, + Message: "this role's insert permissions cannot be enforced on this table", + Retryable: &retryable, + } +} + // typesFailed maps a type layer failure to the request's 503, logging its // cause at most once a minute per tenant and table. func (h *IngestHandler) typesFailed(ctx context.Context, id tenant.ID, table string, err error) *requestAbort { @@ -787,7 +818,9 @@ func writeAbort(w http.ResponseWriter, abort *requestAbort) { if abort.RetryAfter != "" { w.Header().Set("Retry-After", abort.RetryAfter) } - writeJSONErrorBody(w, abort.Status, errorBody{Error: abort.Message, Code: abort.Code, ExceptionCode: abort.ExceptionCode}) + writeJSONErrorBody(w, abort.Status, errorBody{ + Error: abort.Message, Code: abort.Code, ExceptionCode: abort.ExceptionCode, Retryable: abort.Retryable, + }) } // writeMaxBytesError writes a 413 if err is the inbound body-cap overflow and @@ -812,8 +845,9 @@ func writeMaxBytesError(w http.ResponseWriter, err error, limit int64) bool { // // It is not redundant now that the compiled schema answers column policy: a // Defaults entry for a column the shape cannot carry is a COMPILE refusal, so -// without this guard a mis-wired policy would be a 503 naming a ClickHouse -// internal rather than a 403 naming the column the operator has to fix. +// without this guard a mis-wired policy would be a generic 500 (logged with a +// ClickHouse internal) rather than a 403 naming the column the operator has to +// fix. // // Evaluated here rather than per record because the condition is a property of // (table, role, policy) and is identical for every record in the request. Doing @@ -856,12 +890,13 @@ func (h *IngestHandler) policyCheckGuard( reasons = append(reasons, fmt.Sprintf("%q of table %q, which is %s and cannot be inserted", col, table, strings.ToLower(schemaCol.DefaultKind))) case schemaCol.DefaultKind == "EPHEMERAL": - // Insertable, so the row DOES carry a slot for it — but ClickHouse - // never stores an ephemeral column and no query can read one back, so - // the constraint is unverifiable the moment the insert returns. - // Accepting a check that provably does nothing is worse than refusing - // it. An operator wanting this should check the DEFAULT column derived - // from the ephemeral one, which is stored and therefore enforceable. + // A record may supply it — its value feeds the DEFAULT columns over + // it — but ClickHouse never stores it, the published row carries no + // slot for it, and no query can read it back, so the constraint is + // unverifiable the moment the insert returns. Accepting a check that + // provably does nothing is worse than refusing it. An operator + // wanting this should check the DEFAULT column derived from the + // ephemeral one, which is stored and therefore enforceable. slog.ErrorContext(ctx, "policy check references an ephemeral column, which is never stored", "column", col, "table", table, "role", role) reasons = append(reasons, fmt.Sprintf("%q of table %q, which is ephemeral and is never stored", diff --git a/internal/api/ingest_role_test.go b/internal/api/ingest_role_test.go new file mode 100644 index 00000000..cd20a084 --- /dev/null +++ b/internal/api/ingest_role_test.go @@ -0,0 +1,238 @@ +package api + +import ( + "encoding/json" + "log/slog" + "net/http" + "net/http/httptest" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/auth" + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/ingest" + "github.com/Wave-RF/WaveHouse/internal/policy" + "github.com/Wave-RF/WaveHouse/internal/testutil" + "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" +) + +// viewerRaw is a raw ingest request under the viewer role. +func viewerRaw(t *testing.T, table, contentType, body string) *http.Request { + t.Helper() + req := rawIngestRequest(t, table, contentType, body) + return req.WithContext(auth.WithRole(req.Context(), "viewer")) +} + +// viewerInsertPolicy grants the viewer role insert on clicks with ins. +func viewerInsertPolicy(ins *policy.InsertPermissions) PolicySource { + return staticPolicy(&policy.Policy{Tables: map[string]policy.TablePolicy{ + "clicks": {"viewer": {Insert: ins}}, + }}) +} + +// TestIngest_EqCheckOnAColumnTheRoleMayNotWrite: an `_eq` check stamps a column +// the role may not otherwise write — the server-stamped org_id beside an +// allow list of what the client sends. A record omitting it is filled with the +// required value and published with it; a record supplying that value is +// admitted; any other value fails the check; the role's other denied columns +// stay denied. +func TestIngest_EqCheckOnAColumnTheRoleMayNotWrite(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + required := "org-42" + h.PolicySource = viewerInsertPolicy(&policy.InsertPermissions{ + AllowColumns: []string{"page"}, + Check: map[string]policy.Filter{"org_id": {Eq: &required}}, + }) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRaw(t, "clicks", "application/x-ndjson", + `{"page":"/home"}`+"\n"+ + `{"page":"/a","org_id":"org-42"}`+"\n"+ + `{"page":"/b","org_id":"other"}`+"\n"+ + `{"page":"/c","button":"x"}`+"\n"))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.True(t, resultAt(t, resp, 1).Ok, "%+v", resp.Results) + assert.True(t, resultAt(t, resp, 2).Ok, "%+v", resp.Results) + mismatch := resultAt(t, resp, 3) + assert.Equal(t, `check failed for column "org_id"`, mismatch.Error) + assert.Zero(t, mismatch.ExceptionCode, "a failed check is the gateway's verdict, not ClickHouse's") + assert.Equal(t, 117, resultAt(t, resp, 4).ExceptionCode, "button is still denied") + + require.Len(t, pub.Messages, 2) + for _, m := range pub.Messages { + var evt ingest.EventMessage + require.NoError(t, json.Unmarshal(m.Data, &evt)) + assert.Equal(t, []string{"page", "org_id"}, evt.Columns, "the stamped column rides in the row") + assert.Equal(t, "org-42", publishedRow(t, m.Data)["org_id"]) + } +} + +// visitsRegistry is a clicks table whose computed column reads a column a role +// may be denied. +func visitsRegistry(t testing.TB) *discovery.SchemaRegistry { + return testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{{ + Name: "clicks", + Columns: []discovery.Column{ + {Name: "page", Type: "String", Position: 1}, + {Name: "ip", Type: "String", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "''", Position: 2}, + {Name: "ip_hash", Type: "UInt64", HasDefault: true, DefaultKind: "MATERIALIZED", DefaultExpression: "cityHash64(ip)", Position: 3}, + }, + }}) +} + +// TestIngest_DeniedColumnAComputedColumnReads: denying a column that a +// MATERIALIZED expression reads must not take the role's ingest down. The +// denied column stays declared (MATERIALIZED with its own default), so the +// role compiles, a record without it inserts, and one naming it is refused. +func TestIngest_DeniedColumnAComputedColumnReads(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, visitsRegistry(t), pub) + h.PolicySource = viewerInsertPolicy(&policy.InsertPermissions{DenyColumns: []string{"ip"}}) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{"page": "/home"}))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + var evt ingest.EventMessage + require.NoError(t, json.Unmarshal(pub.LastMessage().Data, &evt)) + assert.Equal(t, []string{"page"}, evt.Columns, "the denied column is not on the wire") + + w = httptest.NewRecorder() + h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{"page": "/a", "ip": "1.2.3.4"}))) + require.Equal(t, http.StatusBadRequest, w.Code, "body=%s", w.Body.String()) + msg, code := errorAndCode(t, w) + assert.Equal(t, 117, code) + assert.Contains(t, msg, "ip") + assert.Len(t, pub.Messages, 1) +} + +// TestIngest_RoleShapeRefusalIsNotRetryable: a role whose projection of the +// table does not compile — here, one denied every column, which would publish +// rows with none — is a standing condition of the policy and the schema. It +// answers 500 with retryable:false and no Retry-After, so neither a client nor +// the SDK retries it, and the body names no column or ClickHouse internal. +func TestIngest_RoleShapeRefusalIsNotRetryable(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + h.PolicySource = viewerInsertPolicy(&policy.InsertPermissions{ + DenyColumns: []string{"page", "button", "count", "event_id", "org_id"}, + }) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{"page": "/a"}))) + + require.Equal(t, http.StatusInternalServerError, w.Code, "body=%s", w.Body.String()) + assert.Empty(t, w.Header().Get("Retry-After")) + var body struct { + Error string `json:"error"` + Retryable *bool `json:"retryable"` + } + require.NoError(t, json.Unmarshal(w.Body.Bytes(), &body)) + assert.Equal(t, "this role's insert permissions cannot be enforced on this table", body.Error) + require.NotNil(t, body.Retryable) + assert.False(t, *body.Retryable) + assert.Empty(t, pub.Messages) +} + +// TestIngest_UninjectableCheckValueFailsTheCheck: a check value the column +// cannot hold (`count UInt64` = "abc") does not compile as the column's +// DEFAULT. The role is served without the default, so a record omitting the +// column fails the check (403) rather than every insert being refused, and +// the log says what was done — naming the defaults, not blaming the role. +func TestIngest_UninjectableCheckValueFailsTheCheck(t *testing.T) { + buf := logtest.Capture(t, slog.LevelWarn) + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + required := "abc" + h.PolicySource = viewerInsertPolicy(&policy.InsertPermissions{ + Check: map[string]policy.Filter{"count": {Eq: &required}}, + }) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{"page": "/a"}))) + + require.Equal(t, http.StatusForbidden, w.Code, "body=%s", w.Body.String()) + assert.Equal(t, `check failed for column "count"`, jsonErrorMessage(t, w)) + assert.Empty(t, pub.Messages) + assert.Contains(t, buf.String(), "does not compile with its insert check values as column defaults") +} + +// TestIngest_CheckOnASuppliableEphemeralColumn_StillRefused: a record may now +// supply an EPHEMERAL column a DEFAULT reads, but a check on one still cannot +// be enforced — the value is never stored or published — so it is refused +// with 403 whether the record supplies the column or not, and nothing is +// published. +func TestIngest_CheckOnASuppliableEphemeralColumn_StillRefused(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, ephemeralRegistry(t), pub) + required := "1.2.3.4" + h.PolicySource = viewerInsertPolicy(&policy.InsertPermissions{ + Check: map[string]policy.Filter{"ip": {Eq: &required}}, + }) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(viewerRaw(t, "clicks", "application/x-ndjson", + `{"page":"/a"}`+"\n"+`{"page":"/b","ip":"1.2.3.4"}`+"\n"))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + for i := 1; i <= 2; i++ { + r := resultAt(t, resp, i) + assert.Contains(t, r.Error, "is ephemeral and is never stored", "record %d", i) + assert.Zero(t, r.ExceptionCode, "record %d", i) + } + assert.Empty(t, pub.Messages) + + w = httptest.NewRecorder() + h.Handle(w, withTenant(viewerIngestRequest(t, "clicks", map[string]any{"page": "/c", "ip": "1.2.3.4"}))) + assert.Equal(t, http.StatusForbidden, w.Code, "body=%s", w.Body.String()) + assert.Empty(t, pub.Messages) +} + +// ephemeralRegistry is a clicks table with an EPHEMERAL column a DEFAULT +// reads — the shape EPHEMERAL exists for. +func ephemeralRegistry(t testing.TB) *discovery.SchemaRegistry { + return testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{{ + Name: "clicks", + Columns: []discovery.Column{ + {Name: "page", Type: "String", Position: 1}, + {Name: "ip", Type: "String", HasDefault: true, DefaultKind: "EPHEMERAL", DefaultExpression: "''", Position: 2}, + {Name: "ip_len", Type: "UInt64", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "length(ip)", Position: 3}, + }, + }}) +} + +// TestIngest_EphemeralColumnFeedsItsDefault: a record may name an EPHEMERAL +// column in a format that names its columns; the value feeds the DEFAULT over +// it and is never published. A positional CSV body has no slot for it. +func TestIngest_EphemeralColumnFeedsItsDefault(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, ephemeralRegistry(t), pub) + + for _, tc := range []struct{ contentType, body string }{ + {"application/json", `{"page":"/a","ip":"1.2.3.4"}`}, + {"application/json", `[{"page":"/a","ip":"1.2.3.4"}]`}, + {"application/x-ndjson", `{"page":"/a","ip":"1.2.3.4"}` + "\n"}, + {"text/csv; header=present", "page,ip\n/a,1.2.3.4\n"}, + {"text/tab-separated-values; header=present", "ip\tpage\n1.2.3.4\t/a\n"}, + {"text/csv; header=absent", "/a,7\n"}, + } { + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", tc.contentType, tc.body))) + require.Equal(t, http.StatusOK, w.Code, "%s: body=%s", tc.contentType, w.Body.String()) + + var evt ingest.EventMessage + require.NoError(t, json.Unmarshal(pub.LastMessage().Data, &evt)) + assert.Equal(t, []string{"page", "ip_len"}, evt.Columns, tc.contentType) + assert.JSONEq(t, `["/a", 7]`, string(evt.Row), tc.contentType) + } +} diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index ff8a111a..d500feed 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -256,7 +256,9 @@ func TestIngest_MissingRequiredColumn_TakesTheDefault(t *testing.T) { // columns the exported row actually carries — declaration order minus // MATERIALIZED, ALIAS and EPHEMERAL. The old envelope used InsertableColumns, // which counts EPHEMERAL in, so a table with one announced a column the row did -// not have. Supplying one is the server's own 117. +// not have. raw is an EPHEMERAL column no DEFAULT reads, so a record may not +// supply it: the server's own 117 (see TestIngest_EphemeralColumnFeedsItsDefault +// for one a DEFAULT reads). func TestIngest_ComputedColumns_AreNotOnTheWire(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} @@ -275,7 +277,7 @@ func TestIngest_ComputedColumns_AreNotOnTheWire(t *testing.T) { h.Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/a", "raw": "x"}))) require.Equal(t, http.StatusBadRequest, w.Code) _, code := errorAndCode(t, w) - assert.Equal(t, 117, code, "a record naming an EPHEMERAL column is refused per record, with ClickHouse's code") + assert.Equal(t, 117, code, "a record naming an EPHEMERAL column no DEFAULT reads is refused per record, with ClickHouse's code") } // TestIngest_Batch_PerRecordCodes: a batch reports each refused record's own @@ -449,9 +451,9 @@ func TestIngest_Policy_ColumnDenied(t *testing.T) { h.Handle(w, withTenant(req)) // CONTRACT CHANGE: column policy is answered by compiling the role's own - // schema WITHOUT the denied columns, so the refusal is ClickHouse's - // per-record code 117 — a 400, not the gateway's 403. It no longer confirms - // whether the column exists at all. + // schema with each denied column declared MATERIALIZED, which no INSERT may + // name, so the refusal is ClickHouse's per-record code 117 — a 400, not the + // gateway's 403. It no longer confirms whether the column exists at all. assert.Equal(t, http.StatusBadRequest, w.Code) msg, code := errorAndCode(t, w) assert.Contains(t, msg, "button") @@ -1108,8 +1110,8 @@ func TestIngest_Policy_DenyColumns(t *testing.T) { w := httptest.NewRecorder() h.Handle(w, withTenant(req)) - // CONTRACT CHANGE, as for AllowColumns: a denied column is absent from the - // role's compiled schema, so naming it is ClickHouse's code 117. + // CONTRACT CHANGE, as for AllowColumns: a denied column is one the role's + // compiled schema lets no INSERT name, so naming it is ClickHouse's code 117. assert.Equal(t, http.StatusBadRequest, w.Code) msg, code := errorAndCode(t, w) assert.Contains(t, msg, "count") @@ -3146,12 +3148,12 @@ func computedRegistry(t testing.TB) *discovery.SchemaRegistry { } // TestIngest_CheckOnEphemeralColumn_Rejected: an EPHEMERAL column is the case -// IsInsertable alone does not catch. It IS insertable — the row carries a slot -// for it and the INSERT accepts it — so the check appears enforceable. But -// ClickHouse never stores an ephemeral column and no query can read one back -// (SELECT is code 16, NO_SUCH_COLUMN_IN_TABLE), so the constraint is -// unverifiable the moment the insert returns. A check that provably does -// nothing is worse than a refused one: it reads as tenant isolation and is not. +// IsInsertable alone does not catch. It IS insertable — an INSERT may name it — +// so the check appears enforceable. But ClickHouse never stores an ephemeral +// column and no query can read one back (SELECT is code 16, +// NO_SUCH_COLUMN_IN_TABLE), so the constraint is unverifiable the moment the +// insert returns. A check that provably does nothing is worse than a refused +// one: it reads as tenant isolation and is not. func TestIngest_CheckOnEphemeralColumn_Rejected(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} diff --git a/internal/discovery/discovery.go b/internal/discovery/discovery.go index 2a6eab22..4234f842 100644 --- a/internal/discovery/discovery.go +++ b/internal/discovery/discovery.go @@ -45,8 +45,8 @@ type Column struct { HasDefault bool `json:"has_default"` // DefaultKind is how the column's default is declared, verbatim from // system.columns.default_kind: "" (none), "DEFAULT", "MATERIALIZED", - // "ALIAS", or "EPHEMERAL". It is what decides whether a record may carry a - // value for the column — see IsInsertable. HasDefault is the boolean + // "ALIAS", or "EPHEMERAL". It is what decides whether an INSERT may name the + // column — see IsInsertable. HasDefault is the boolean // reading of the same field, so the two never disagree about whether a // default exists. DefaultKind string `json:"default_kind,omitempty"` @@ -81,15 +81,6 @@ type TableSchema struct { // system.tables by the time the second query ran — the two scans are not one // snapshot, so a consumer must not treat an empty DDL as "no such table". DDL string `json:"-"` - - // insertable/insertableNames memoize InsertableColumns/InsertableColumnNames, - // which are per-table constants that the ingest path would otherwise rebuild - // once per record — on a 20-column table that is ~2 KB of garbage per record. - // Filled once by cacheInsertable when the registry builds the schema, and - // never written again, so concurrent readers need no lock. A TableSchema - // built as a literal (tests) leaves them nil and takes the uncached path. - insertable []Column - insertableNames []string } // ColumnNames returns the table's column names in their discovered order @@ -99,8 +90,8 @@ type TableSchema struct { // schema with no columns. func (ts *TableSchema) ColumnNames() []string { return columnNames(ts.Columns) } -// IsInsertable reports whether a record may carry a value for this column — -// whether naming it in an INSERT's column list is legal. +// IsInsertable reports whether naming this column in an INSERT's column list +// is legal. // // ClickHouse refuses exactly two kinds, verified against a live server // (26.6.3): a MATERIALIZED column is `Cannot insert column …, because it is @@ -109,6 +100,11 @@ func (ts *TableSchema) ColumnNames() []string { return columnNames(ts.Columns) } // no storage to write. Everything else takes a value: a plain column, a // DEFAULT column, and an EPHEMERAL one, which exists precisely to be inserted // into (it is insert-only — never stored, never selected). +// +// It is not what an ingested record may carry, nor what a published row +// holds: both are the type layer's (typelayer.Table.WireColumns, and the +// EPHEMERAL columns it lets a record supply), and an EPHEMERAL column is +// never on the wire. func (c Column) IsInsertable() bool { switch c.DefaultKind { case "MATERIALIZED", "ALIAS": @@ -118,43 +114,7 @@ func (c Column) IsInsertable() bool { } } -// InsertableColumns returns the columns a record may carry, in declaration -// order — the positional contract for a row on the ingest path. Computed -// columns are left out because naming one in an INSERT is an error, not -// because they are uninteresting: they stay in Columns, so the schema endpoint -// and the query path still see the whole table. -func (ts *TableSchema) InsertableColumns() []Column { - if ts.insertable != nil { - return ts.insertable - } - return computeInsertable(ts.Columns) -} - -// cacheInsertable fills the memoized subsets. Called once per table per refresh, -// before the schema is published to readers. -func (ts *TableSchema) cacheInsertable() { - ts.insertable = computeInsertable(ts.Columns) - ts.insertableNames = columnNames(ts.insertable) -} - -// computeInsertable filters to the columns a record may carry, preserving -// declaration order — the positional contract the wire envelope depends on. -func computeInsertable(cols []Column) []Column { - out := make([]Column, 0, len(cols)) - for _, c := range cols { - if c.IsInsertable() { - out = append(out, c) - } - } - // Cap the slice to its length. The memo is handed to every caller by - // reference, and on a table with a computed column cap > len — so an append - // by some future caller would write into the shared, concurrently-read - // backing array instead of copying. Capping forces that append to allocate. - return out[:len(out):len(out)] -} - -// columnNames projects a column slice to its names, shared by ColumnNames and -// InsertableColumnNames so the two cannot drift. +// columnNames projects a column slice to its names. func columnNames(cols []Column) []string { names := make([]string, 0, len(cols)) for _, c := range cols { @@ -170,7 +130,7 @@ func columnNames(cols []Column) []string { // // It returns the Column rather than a bool because "does the table have it" is // rarely the whole question — a caller on the ingest path also has to know -// whether a record may carry a value for it (IsInsertable). +// whether an INSERT may name it (IsInsertable). func (ts *TableSchema) Lookup(name string) (Column, bool) { for _, c := range ts.Columns { if c.Name == name { @@ -180,15 +140,6 @@ func (ts *TableSchema) Lookup(name string) (Column, bool) { return Column{}, false } -// InsertableColumnNames is InsertableColumns reduced to names, for the wire -// envelope's column list. Returns an empty (non-nil) slice when none qualify. -func (ts *TableSchema) InsertableColumnNames() []string { - if ts.insertableNames != nil { - return ts.insertableNames - } - return columnNames(ts.InsertableColumns()) -} - // SchemaRegistry discovers and caches one tenant's ClickHouse table schemas. type SchemaRegistry struct { // source supplies the tenant's connection and the database to discover @@ -378,7 +329,6 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { var noDDL []string published := make([]*TableSchema, 0, len(tables)) for _, ts := range tables { - ts.cacheInsertable() published = append(published, ts) if ts.DDL == "" { noDDL = append(noDDL, ts.Name) diff --git a/internal/discovery/discovery_test.go b/internal/discovery/discovery_test.go index 601f75ce..c3607fc6 100644 --- a/internal/discovery/discovery_test.go +++ b/internal/discovery/discovery_test.go @@ -487,7 +487,7 @@ func TestOnRefresh_FiresAfterSwapWithPublishedSchemas(t *testing.T) { assert.Equal(t, "Europe/Berlin", gotTZ) require.Len(t, gotTables, 1) assert.Equal(t, "events", gotTables[0].Name) - assert.Equal(t, []string{"id"}, gotTables[0].InsertableColumnNames(), "hook sees the memoized schema") + assert.Equal(t, []string{"id"}, gotTables[0].ColumnNames(), "hook sees the published schema") } // TestOnRefresh_RunsBeforeLoaded: "loaded" must imply "bound", so on the first @@ -1113,36 +1113,6 @@ func TestColumn_IsInsertable(t *testing.T) { } } -// TestTableSchema_InsertableColumns: computed columns are dropped from the -// ingest contract but stay in Columns — the schema endpoint and the query path -// still see the whole table. -func TestTableSchema_InsertableColumns(t *testing.T) { - t.Parallel() - ts := &TableSchema{Name: "t", Columns: []Column{ - {Name: "id", Position: 1}, - {Name: "mat", DefaultKind: "MATERIALIZED", Position: 2}, - {Name: "page", DefaultKind: "DEFAULT", Position: 3}, - {Name: "ali", DefaultKind: "ALIAS", Position: 4}, - }} - assert.Equal(t, []string{"id", "page"}, ts.InsertableColumnNames()) - assert.Equal(t, []string{"id", "mat", "page", "ali"}, ts.ColumnNames(), - "the full column list is untouched") - - got := ts.InsertableColumns() - require.Len(t, got, 2) - assert.Equal(t, uint64(3), got[1].Position, "declaration order and ordinals survive the filter") -} - -// TestTableSchema_InsertableColumnNames_NoneQualify: a table of nothing but -// computed columns yields an empty, non-nil list rather than nil. -func TestTableSchema_InsertableColumnNames_NoneQualify(t *testing.T) { - t.Parallel() - ts := &TableSchema{Name: "t", Columns: []Column{{Name: "ali", DefaultKind: "ALIAS"}}} - got := ts.InsertableColumnNames() - assert.Empty(t, got) - assert.NotNil(t, got) -} - // TestRefresh_CapturesDefaultKind: the kind is what decides insertability, so // it must survive the scan verbatim rather than being flattened to HasDefault. func TestRefresh_CapturesDefaultKind(t *testing.T) { @@ -1161,38 +1131,6 @@ func TestRefresh_CapturesDefaultKind(t *testing.T) { assert.Equal(t, "MATERIALIZED", ts.Columns[1].DefaultKind) assert.True(t, ts.Columns[1].HasDefault, "a computed column still reads as having a default") assert.Equal(t, "DEFAULT", ts.Columns[2].DefaultKind) - assert.Equal(t, []string{"id", "page"}, ts.InsertableColumnNames()) -} - -func TestInsertableColumns_CachedAndUncachedAgree(t *testing.T) { - t.Parallel() - // A literal-built TableSchema (what tests construct) leaves the memo nil and - // takes the compute path; cacheInsertable is what the registry calls after a - // refresh. Both must answer identically, or a discovered table and a - // hand-built one disagree about what may be inserted. - ts := &TableSchema{Name: "clicks", Columns: []Column{ - {Name: "id", Type: "UInt64"}, - {Name: "day", Type: "Date", DefaultKind: "MATERIALIZED"}, - {Name: "page", Type: "String"}, - {Name: "alias", Type: "String", DefaultKind: "ALIAS"}, - {Name: "note", Type: "String", DefaultKind: "DEFAULT"}, - }} - - uncached := ts.InsertableColumns() - uncachedNames := ts.InsertableColumnNames() - require.Nil(t, ts.insertable, "reading must not populate the memo") - - ts.cacheInsertable() - assert.Equal(t, uncached, ts.InsertableColumns()) - assert.Equal(t, uncachedNames, ts.InsertableColumnNames()) - assert.Equal(t, []string{"id", "page", "note"}, ts.InsertableColumnNames(), - "MATERIALIZED and ALIAS are excluded; DEFAULT and plain are not") - - // The memo is returned by reference, so every caller shares one slice. That - // is safe only while no caller writes to it — pin the contract here, since a - // caller that appended would corrupt every later request for this table. - first := ts.InsertableColumns() - second := ts.InsertableColumns() - require.NotEmpty(t, first) - assert.Same(t, &first[0], &second[0], "callers share the memoized backing array") + assert.False(t, ts.Columns[1].IsInsertable()) + assert.True(t, ts.Columns[2].IsInsertable()) } diff --git a/internal/stream/hub.go b/internal/stream/hub.go index 3915c32e..bc16478c 100644 --- a/internal/stream/hub.go +++ b/internal/stream/hub.go @@ -12,6 +12,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/typelayer" ) // Hub fans live events out to SSE subscribers. Column projection is serialized ONCE @@ -521,12 +522,13 @@ func (h *Hub) SubscribeSchemaFrame(id tenant.ID, table, role string, sub *Subscr return Frame{}, false } } - // The insertable subset, matching what a full-width envelope carries — a - // computed column never appears in a published row, so announcing it here - // would guarantee a drift re-announcement on the very first event. An - // envelope from a role that may write fewer columns carries fewer, and is - // announced again when it arrives, like any other change of list. - _, projected := projectIndices(schema.InsertableColumnNames(), perms) + // The wire columns, exactly what a full-width envelope carries — a + // MATERIALIZED, ALIAS or EPHEMERAL column never appears in a published + // row, so announcing one here would guarantee a drift re-announcement on + // the very first event. An envelope from a role that may write fewer + // columns carries fewer, and is announced again when it arrives, like any + // other change of list. + _, projected := projectIndices(typelayer.WireColumnsOf(schema), perms) sig := schemaSignature(table, projected) if !sub.needsSchema(sig) { return Frame{}, false // already announced (a live event beat us here) diff --git a/internal/stream/hub_test.go b/internal/stream/hub_test.go index 9fe46acf..7eadf22a 100644 --- a/internal/stream/hub_test.go +++ b/internal/stream/hub_test.go @@ -1586,28 +1586,34 @@ func TestHub_SchemaFrame_DroppedAnnouncementDropsItsRow(t *testing.T) { // TestHub_SubscribeSchemaFrame_ExcludesComputedColumns: the connect-time // announcement comes from the registry while every event's list comes from the -// envelope, which carries only insertable columns. If the two disagreed, the -// very first event would force a pointless drift re-announcement. +// envelope, which carries the type layer's wire columns — no MATERIALIZED, +// ALIAS or EPHEMERAL column. If the two disagreed, the very first event would +// force a pointless drift re-announcement. The envelope side is read off a +// compiled handle, as ingest builds it, not written down here. func TestHub_SubscribeSchemaFrame_ExcludesComputedColumns(t *testing.T) { t.Parallel() - reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ - {Name: "clicks", Columns: []discovery.Column{ - {Name: "page", Type: "String"}, - {Name: "digest", Type: "String", DefaultKind: "MATERIALIZED", HasDefault: true}, - {Name: "country", Type: "String"}, - }}, - }) - hub := NewHub(nil, fixedRegistry(reg), nil) + table := &discovery.TableSchema{Name: "clicks", Columns: []discovery.Column{ + {Name: "page", Type: "String", Position: 1}, + {Name: "digest", Type: "String", DefaultKind: "MATERIALIZED", DefaultExpression: "upper(page)", HasDefault: true, Position: 2}, + {Name: "raw", Type: "String", DefaultKind: "EPHEMERAL", HasDefault: true, Position: 3}, + {Name: "country", Type: "String", Position: 4}, + }} + tbl, err := typelayertest.TestEngine(t, table).Table(tenant.Default, "clicks") + require.NoError(t, err) + envelope := tbl.WireColumns + tbl.Release() + hub := NewHub(nil, fixedRegistry(testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{table})), nil) sub := NewSubscriber(nil, nil) f, ok := hub.SubscribeSchemaFrame(tenant.Default, "clicks", "public", sub) require.True(t, ok) cols := frameColumns(t, f) + assert.Equal(t, envelope, cols) assert.Equal(t, []string{"page", "country"}, cols) // The first event announces nothing new, because the lists agree. hub.Add(topicOf("clicks"), "public", sub) - hub.Broadcast(topicOf("clicks"), rawEventCols(t, "clicks", "t1", cols, + hub.Broadcast(topicOf("clicks"), rawEventCols(t, "clicks", "t1", envelope, map[string]any{"page": "/a", "country": "US"})) _, row := recvEventCols(t, sub, cols) assert.Equal(t, "/a", row["page"]) diff --git a/internal/typelayer/errors.go b/internal/typelayer/errors.go index d8292f39..2285638d 100644 --- a/internal/typelayer/errors.go +++ b/internal/typelayer/errors.go @@ -38,6 +38,25 @@ func IsUnavailable(err error) bool { return errors.As(err, &u) } +// RoleRefused reports that a role's projection of a table does not compile, +// while the table itself does. Only Engine.RoleTable returns it. +// +// Unlike Unavailable it is a standing condition: it follows from the role's +// insert policy and the table's schema, and the refusal is cached for the +// schema generation, so the same request fails the same way until an operator +// changes one of them. A caller must not answer it with a retry hint. +type RoleRefused struct { + Tenant tenant.ID + Table string + // Cause is the operator-facing reason: ClickHouse's own compile refusal, + // or the shape the role asked for that cannot be expressed. + Cause string +} + +func (e *RoleRefused) Error() string { + return fmt.Sprintf("chtypes cannot compile the role's schema for tenant %q table %q: %s", e.Tenant, e.Table, e.Cause) +} + // ErrColumnsDrift is returned by ParseRow when the envelope's column list is // not an INSERT column list the current compiled handle accepts: it names a // column the handle does not export, or names one twice. The event predates a diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go index 439d751c..105865f1 100644 --- a/internal/typelayer/ingest.go +++ b/internal/typelayer/ingest.go @@ -153,13 +153,17 @@ func (t *Table) IngestWith(format Format, opts IngestOptions, body []byte, check // another handle rejects the whole call. s := t.pool.acquire() defer t.pool.release(s) + var columns []string + if format != FormatCSV && format != FormatTSV { + columns = t.inputs // see inputColumns; positional formats map to WireColumns + } filter, uniform := t.checkFilter(s, checks) - res, err := export(s, format, body, settings, filter) + res, err := export(s, format, body, settings, columns, filter) if err != nil && filter != nil { // The cached filter was evicted and closed between lookup and use. Fail // the checks closed, as an evaluation error would, not the request. filter, uniform = nil, ReasonDecline - res, err = export(s, format, body, settings, nil) + res, err = export(s, format, body, settings, columns, nil) } if err != nil { return Batch{}, err @@ -215,12 +219,17 @@ func (t *Table) checkFilter(s *schemaSlot, checks []Predicate) (*chtypes.LoadedF return nil, ReasonDecline } -// export is the one parse. With no filter RowsExportWith is RowsExport. -func export(s *schemaSlot, format Format, body []byte, settings map[string]string, f *chtypes.LoadedFilter) (chtypes.BatchResult, error) { - if f == nil { - return s.schema.RowsExportWith(format, body, settings, chtypes.JSONCompactEachRow) +// export is the one parse, with columns as the INSERT column list (nil: none). +// With no filter RowsExportWith is RowsExport. +func export(s *schemaSlot, format Format, body []byte, settings map[string]string, columns []string, f *chtypes.LoadedFilter) (chtypes.BatchResult, error) { + opts := make([]chtypes.RowsOption, 0, 2) + if columns != nil { + opts = append(opts, chtypes.WithColumns(columns)) + } + if f != nil { + opts = append(opts, chtypes.WithRowFilter(f)) } - return s.schema.RowsExportWith(format, body, settings, chtypes.JSONCompactEachRow, chtypes.WithRowFilter(f)) + return s.schema.RowsExportWith(format, body, settings, chtypes.JSONCompactEachRow, opts...) } // rowVerdict maps one chtypes RowResult. An unsupported setting is the engine diff --git a/internal/typelayer/ingest_test.go b/internal/typelayer/ingest_test.go index 276cd21c..0980d09a 100644 --- a/internal/typelayer/ingest_test.go +++ b/internal/typelayer/ingest_test.go @@ -93,8 +93,10 @@ func TestIngest_OverflowIsStoredTruth(t *testing.T) { } // TestIngest_ComputedColumnsAreRejectedPerRecord: a record naming a -// MATERIALIZED, ALIAS or EPHEMERAL column gets ClickHouse's own 117 and the -// rest of the batch still gets verdicts — no WaveHouse-side guard needed. +// MATERIALIZED or ALIAS column gets ClickHouse's own 117 and the rest of the +// batch still gets verdicts — no WaveHouse-side guard needed. An EPHEMERAL +// column is the one non-stored kind a record may name: its value feeds the +// DEFAULT over it and is never exported. func TestIngest_ComputedColumnsAreRejectedPerRecord(t *testing.T) { schema := &discovery.TableSchema{ Name: "computed", @@ -103,6 +105,7 @@ func TestIngest_ComputedColumnsAreRejectedPerRecord(t *testing.T) { {Name: "e", Type: "UInt8", DefaultKind: "EPHEMERAL", HasDefault: true, Position: 2}, {Name: "d", Type: "UInt8", DefaultKind: "DEFAULT", DefaultExpression: "e + 1", HasDefault: true, Position: 3}, {Name: "a", Type: "UInt8", DefaultKind: "ALIAS", DefaultExpression: "id + 2", HasDefault: true, Position: 4}, + {Name: "m", Type: "UInt32", DefaultKind: "MATERIALIZED", DefaultExpression: "id * 2", HasDefault: true, Position: 5}, }, } eng := testEngine(t, schema) @@ -117,20 +120,107 @@ func TestIngest_ComputedColumnsAreRejectedPerRecord(t *testing.T) { `{"id":2,"e":5}`, `{"id":3,"a":9}`, `{"id":4,"d":7}`, + `{"id":5,"m":1}`, }, "\n") + "\n" batch, err := tbl.Ingest(FormatJSONEachRow, []byte(body)) require.NoError(t, err) - require.Len(t, batch.Rows, 4) + require.Len(t, batch.Rows, 5) - assert.True(t, batch.Rows[0].Accepted) - assert.Equal(t, 117, batch.Rows[1].Code) - assert.Contains(t, batch.Rows[1].Message, "e") + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, "[1, 1]", string(batch.Rows[0].Line), "an absent EPHEMERAL takes its own default") + require.True(t, batch.Rows[1].Accepted, batch.Rows[1].Message) + assert.Equal(t, "[2, 6]", string(batch.Rows[1].Line), "the EPHEMERAL value feeds the DEFAULT and is not exported") assert.Equal(t, 117, batch.Rows[2].Code) assert.Contains(t, batch.Rows[2].Message, "a") - assert.True(t, batch.Rows[3].Accepted) + require.True(t, batch.Rows[3].Accepted, batch.Rows[3].Message) // ClickHouse's writer separates cells with ", " — the worker inserts these // bytes verbatim, so nothing may re-render them. assert.Equal(t, "[4, 7]", string(batch.Rows[3].Line)) + assert.Equal(t, 117, batch.Rows[4].Code) + assert.Contains(t, batch.Rows[4].Message, "m") +} + +// TestIngest_EphemeralInputFollowsTheFormat: the formats that name their +// columns accept an EPHEMERAL one; positional CSV/TSV map their fields to the +// wire columns, so an ephemeral slot is not one of them. +func TestIngest_EphemeralInputFollowsTheFormat(t *testing.T) { + schema := &discovery.TableSchema{ + Name: "eph", + Columns: []discovery.Column{ + {Name: "page", Type: "String", Position: 1}, + {Name: "ip", Type: "String", DefaultKind: "EPHEMERAL", DefaultExpression: "''", HasDefault: true, Position: 2}, + {Name: "ip_len", Type: "UInt64", DefaultKind: "DEFAULT", DefaultExpression: "length(ip)", HasDefault: true, Position: 3}, + }, + } + eng := testEngine(t, schema) + tbl, err := eng.Table(tenant.Default, "eph") + require.NoError(t, err) + defer tbl.Release() + require.Equal(t, []string{"page", "ip_len"}, tbl.WireColumns) + + for _, tc := range []struct { + name string + format Format + body string + want string + }{ + {"JSONEachRow", FormatJSONEachRow, `{"page":"/a","ip":"1.2.3.4"}` + "\n", `["\/a", 7]`}, + {"CSVWithNames", FormatCSVWithNames, "page,ip\n/a,1.2.3.4\n", `["\/a", 7]`}, + {"TSVWithNames", FormatTSVWithNames, "ip\tpage\n1.2.3.4\t/a\n", `["\/a", 7]`}, + {"CSV is the wire columns", FormatCSV, "/a,3\n", `["\/a", 3]`}, + {"TSV is the wire columns", FormatTSV, "/a\t3\n", `["\/a", 3]`}, + } { + t.Run(tc.name, func(t *testing.T) { + batch, err := tbl.Ingest(tc.format, []byte(tc.body)) + require.NoError(t, err) + require.Nil(t, batch.Refused) + require.Len(t, batch.Rows, 1) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, tc.want, string(batch.Rows[0].Line)) + }) + } + + // A positional body has no slot for the ephemeral column: a third field is + // an error, not ip. + batch, err := tbl.IngestWith(FormatCSV, IngestOptions{StrictPositional: true}, []byte("/a,1.2.3.4,7\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 1) + assert.False(t, batch.Rows[0].Accepted) +} + +// TestIngest_EphemeralAServerExpressionReadsIsRefused: the server computes a +// MATERIALIZED column itself from the inserted row, which never carries an +// ephemeral value, so an EPHEMERAL column one reads would be silently ignored +// there — even though a DEFAULT reads it too. It is refused instead (117), +// while another EPHEMERAL column of the same table that only a DEFAULT reads +// is still accepted. +func TestIngest_EphemeralAServerExpressionReadsIsRefused(t *testing.T) { + schema := &discovery.TableSchema{ + Name: "eph", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt32", Position: 1}, + {Name: "e", Type: "UInt8", DefaultKind: "EPHEMERAL", HasDefault: true, Position: 2}, + {Name: "m", Type: "UInt16", DefaultKind: "MATERIALIZED", DefaultExpression: "e * 2", HasDefault: true, Position: 3}, + {Name: "de", Type: "UInt16", DefaultKind: "DEFAULT", DefaultExpression: "e + 1", HasDefault: true, Position: 4}, + {Name: "f", Type: "UInt8", DefaultKind: "EPHEMERAL", HasDefault: true, Position: 5}, + {Name: "d", Type: "UInt16", DefaultKind: "DEFAULT", DefaultExpression: "f + 1", HasDefault: true, Position: 6}, + {Name: "md", Type: "UInt16", DefaultKind: "MATERIALIZED", DefaultExpression: "d * 2", HasDefault: true, Position: 7}, + }, + } + eng := testEngine(t, schema) + tbl, err := eng.Table(tenant.Default, "eph") + require.NoError(t, err) + defer tbl.Release() + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":1,"e":5}`+"\n"+`{"id":2,"f":5}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 2) + assert.Equal(t, 117, batch.Rows[0].Code, "e feeds a MATERIALIZED column the server computes without it") + assert.Contains(t, batch.Rows[0].Message, "e") + // md reads d, which travels on the wire already computed from f, so the + // server's md agrees with this one: f is accepted. + require.True(t, batch.Rows[1].Accepted, batch.Rows[1].Message) + assert.Equal(t, "[2, 1, 6]", string(batch.Rows[1].Line)) } // TestInsertSettings_AreAllSupported: every setting typelayer passes must be one @@ -277,3 +367,52 @@ func TestIngest_TypeGatedColumnsInsertAndFilter(t *testing.T) { assert.Contains(t, []int{44, 455}, res.ErrCode, "%s without the gates: %s", ts.Type, res.ErrMsg) } } + +// TestIngest_EphemeralNoDefaultReadsStaysUnlisted: an EPHEMERAL column is +// listed only when a DEFAULT reads it. Listing one nothing reads makes the +// artifact reject, with no code, every record that omits it in a table that +// computes any column, so it stays refused (117) — its value would change +// nothing — and records omitting the listed ones still get verdicts. +func TestIngest_EphemeralNoDefaultReadsStaysUnlisted(t *testing.T) { + schema := &discovery.TableSchema{ + Name: "eph", + Columns: []discovery.Column{ + {Name: "id", Type: "UInt32", Position: 1}, + {Name: "e1", Type: "UInt8", DefaultKind: "EPHEMERAL", HasDefault: true, Position: 2}, + {Name: "e2", Type: "String", DefaultKind: "EPHEMERAL", DefaultExpression: "'zz'", HasDefault: true, Position: 3}, + {Name: "unread", Type: "String", DefaultKind: "EPHEMERAL", DefaultExpression: "''", HasDefault: true, Position: 4}, + {Name: "d1", Type: "UInt16", DefaultKind: "DEFAULT", DefaultExpression: "e1 + 1", HasDefault: true, Position: 5}, + {Name: "d2", Type: "UInt64", DefaultKind: "DEFAULT", DefaultExpression: "length(e2)", HasDefault: true, Position: 6}, + {Name: "m", Type: "UInt64", DefaultKind: "MATERIALIZED", DefaultExpression: "id * 2", HasDefault: true, Position: 7}, + }, + } + eng := testEngine(t, schema) + tbl, err := eng.Table(tenant.Default, "eph") + require.NoError(t, err) + defer tbl.Release() + require.Equal(t, []string{"id", "d1", "d2"}, tbl.WireColumns) + + body := strings.Join([]string{ + `{"id":1}`, + `{"id":2,"e1":5}`, + `{"id":3,"e2":"abc"}`, + `{"id":4,"e1":5,"e2":"abc"}`, + `{"id":5,"unread":"x"}`, + }, "\n") + "\n" + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(body)) + require.NoError(t, err) + require.Len(t, batch.Rows, 5) + for i, want := range []string{"[1, 1, 2]", "[2, 6, 2]", "[3, 1, 3]", "[4, 6, 3]"} { + require.True(t, batch.Rows[i].Accepted, "record %d: %s", i+1, batch.Rows[i].Message) + assert.Equal(t, want, string(batch.Rows[i].Line), "record %d", i+1) + } + assert.Equal(t, 117, batch.Rows[4].Code) + assert.Contains(t, batch.Rows[4].Message, "unread") + + // A header that leaves a listed EPHEMERAL column out is the same omission. + batch, err = tbl.Ingest(FormatCSVWithNames, []byte("id,e1\n6,5\n")) + require.NoError(t, err) + require.Nil(t, batch.Refused) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, "[6, 6, 2]", string(batch.Rows[0].Line)) +} diff --git a/internal/typelayer/roletable.go b/internal/typelayer/roletable.go index 145de933..f179d24b 100644 --- a/internal/typelayer/roletable.go +++ b/internal/typelayer/roletable.go @@ -3,6 +3,7 @@ package typelayer import ( "container/list" "crypto/sha256" + "errors" "fmt" "log/slog" "maps" @@ -28,21 +29,25 @@ const roleCacheSize = 256 // the same shape share one compiled handle. // // Columns is the set of columns the role may write. nil means "every column" -// and is the identity shape; a non-nil, EMPTY slice means "no column", which -// compiles to nothing and fails closed. Order is irrelevant — the DDL always -// follows the table's own declaration order — and a name the table does not -// have is ignored. Columns the role cannot supply anyway (MATERIALIZED, ALIAS, -// EPHEMERAL) are always kept: dropping one would change what the server -// computes, and a MATERIALIZED expression over a dropped column would not -// compile at all. +// and is the identity shape. Order is irrelevant — the DDL always follows the +// table's own declaration order — and a name the table does not have is +// ignored. A column the role may not write stays declared, re-declared +// MATERIALIZED with what the server stores when an INSERT omits it (its own +// DEFAULT expression, or its type's default): a record naming it is then +// refused like any column no INSERT may name (117), it is never on the wire, +// and every expression reading it still compiles and sees the value the +// server will store. A shape that leaves the role no plain or DEFAULT column +// to write is refused: it would publish rows with no columns. MATERIALIZED +// and ALIAS columns are declared as the table declares them; an EPHEMERAL one +// too, and a record may supply it only if the role may write it (and only +// where inputColumns lists it). // // Defaults maps a column to a literal value injected when a record omits it, // rendered into the column's DEFAULT clause. A value the record DOES supply // still wins (measured on the 26.6 and 26.8 artifacts; see // TestRoleTable_DefaultInjectsWhenAbsentAndLosesToASuppliedValue). Every key -// must name a column the shape -// keeps and must be an ordinary column (no DEFAULT, or a plain DEFAULT) — -// anything else is a programming error and returns an error rather than +// must name a column the role may write and must be an ordinary column (no +// DEFAULT, or a plain DEFAULT) — anything else is refused rather than // silently reshaping the table. type RoleShape struct { Columns []string @@ -55,6 +60,11 @@ func (s RoleShape) identity() bool { return s.Columns == nil && len(s.Defaults) == 0 } +// writable reports whether the role may supply column name. +func (s RoleShape) writable(name string) bool { + return s.Columns == nil || slices.Contains(s.Columns, name) +} + // key is the cache key: the generation that owns the schema plus a canonical // hash of the shape. Every component is length-prefixed, so no column name or // claim value can spell another shape's encoding. @@ -87,8 +97,8 @@ func (s RoleShape) key(generation uint64) string { // handle stays alive and stable until it does. // // The shape is answered by ClickHouse's own parser rather than by a Go walk -// over the record's keys: a column the role may not write is -// simply absent from the compiled DDL, so a record naming it is refused +// over the record's keys: a column the role may not write is declared +// MATERIALIZED, which no INSERT may name, so a record naming it is refused // per-row with ClickHouse's own code 117 "Unknown field found while parsing // JSONEachRow format: x"; a Defaults column is declared DEFAULT '', // quoted by the library's own QuoteLiteral, so an absent value is filled and a @@ -98,8 +108,9 @@ func (s RoleShape) key(generation uint64) string { // table itself — no second handle, no cache entry. // // A shape that does not compile is cached as a negative entry and reported as -// *Unavailable, so a broken policy costs one compile and one log line per -// generation rather than one per request. +// *RoleRefused, so a broken policy costs one compile and one log line per +// generation rather than one per request. An error from resolving the base +// table itself is Table's *Unavailable. func (e *Engine) RoleTable(id tenant.ID, table string, shape RoleShape) (*Table, error) { base, err := e.Table(id, table) if err != nil { @@ -134,7 +145,7 @@ func (t *Table) roleTable(shape RoleShape) (*Table, *Table, error) { c.order.MoveToFront(el) e := el.Value.(*roleEntry) if e.table == nil { - return nil, nil, &Unavailable{Tenant: t.tenant, Table: t.Name, Cause: e.cause} + return nil, nil, &RoleRefused{Tenant: t.tenant, Table: t.Name, Cause: e.cause} } // Taken while the base read lock is still held, so a rebind cannot be // closing this projection underneath us. @@ -156,7 +167,7 @@ func (t *Table) roleTable(shape RoleShape) (*Table, *Table, error) { evicted = c.evictOldestLocked() } if rt == nil { - return nil, evicted, &Unavailable{Tenant: t.tenant, Table: t.Name, Cause: cause} + return nil, evicted, &RoleRefused{Tenant: t.tenant, Table: t.Name, Cause: cause} } rt.mu.RLock() return rt, evicted, nil @@ -183,79 +194,101 @@ func (t *Table) compileRole(shape RoleShape) (*Table, string) { p.close() return nil, cause } + wire = deriveWireColumns(schema, wire) return &Table{ Name: t.Name, Generation: t.Generation, - WireColumns: deriveWireColumns(schema, wire), + WireColumns: wire, tenant: t.tenant, pool: p, cols: declared, + inputs: inputColumns(t.lib, cols, wire, shape.writable), lib: t.lib, }, "" } // roleColumns projects the table's discovered columns onto a shape. It returns -// the declaration list and the wire column names (the declaration list minus -// the three kinds a positional INSERT never carries). An injected value is +// the declaration list and the wire column names (the plain and DEFAULT +// columns the role may write — what RowsExport emits). An injected value is // quoted by lib.QuoteLiteral, ClickHouse's own quoteString, so it reaches the // compiler as one string literal whatever bytes it holds. +// +// A column the role may not write is re-declared MATERIALIZED rather than +// dropped. Dropped, every DEFAULT, MATERIALIZED or ALIAS expression reading it +// would stop compiling, and the whole role would be refused for a column it +// never asked to write. MATERIALIZED with what the server stores when the +// worker's INSERT omits it keeps it off the wire and unnamable (117), and +// keeps what those expressions compute here equal to what the server will +// store. func roleColumns(lib *chtypes.Library, src []discovery.Column, shape RoleShape) ([]chtypes.DiscoveredColumn, []string, error) { - var allowed map[string]struct{} - if shape.Columns != nil { - allowed = make(map[string]struct{}, len(shape.Columns)) - for _, c := range shape.Columns { - allowed[c] = struct{}{} - } - } - cols := make([]chtypes.DiscoveredColumn, 0, len(src)) wire := make([]string, 0, len(src)) - kept := make(map[string]discovery.Column, len(src)) + declared := make(map[string]struct{}, len(src)) for _, c := range src { - computed := c.DefaultKind == "MATERIALIZED" || c.DefaultKind == "ALIAS" || c.DefaultKind == "EPHEMERAL" - if allowed != nil && !computed { - if _, ok := allowed[c.Name]; !ok { - continue - } - } - dc := chtypes.DiscoveredColumn{ - Name: c.Name, - Type: c.Type, - DefaultKind: c.DefaultKind, - DefaultExpression: c.DefaultExpression, - Position: c.Position, - } - if v, inject := shape.Defaults[c.Name]; inject { - if computed { + declared[c.Name] = struct{}{} + dc := discoveredColumn(c) + v, inject := shape.Defaults[c.Name] + switch { + case c.DefaultKind == "MATERIALIZED" || c.DefaultKind == "ALIAS" || c.DefaultKind == "EPHEMERAL": + if inject { return nil, nil, fmt.Errorf( "cannot inject a default into column %q: it is %s", c.Name, c.DefaultKind) } - lit, err := lib.QuoteLiteral(v) + case !shape.writable(c.Name): + // A default the role may not write cannot be expressed, and + // dropping it would turn "force this value" into "whatever the + // server defaults to". Fail loudly instead. + if inject { + return nil, nil, fmt.Errorf( + "cannot inject a default into column %q: the role may not write it", c.Name) + } + expr, err := storedDefault(lib, c) if err != nil { - return nil, nil, fmt.Errorf("cannot quote the default for column %q: %w", c.Name, err) + return nil, nil, err + } + dc.DefaultKind, dc.DefaultExpression = "MATERIALIZED", expr + default: + if inject { + lit, err := lib.QuoteLiteral(v) + if err != nil { + return nil, nil, fmt.Errorf("cannot quote the default for column %q: %w", c.Name, err) + } + dc.DefaultKind, dc.DefaultExpression = "DEFAULT", lit } - dc.DefaultKind, dc.DefaultExpression = "DEFAULT", lit - } - cols = append(cols, dc) - if !computed { wire = append(wire, c.Name) } - kept[c.Name] = c + cols = append(cols, dc) } - // A default for a column this shape does not carry cannot be expressed, - // and silently dropping it would turn "force this value" into "whatever - // the caller sent". Fail loudly instead. for _, name := range slices.Sorted(maps.Keys(shape.Defaults)) { - if _, ok := kept[name]; !ok { + if _, ok := declared[name]; !ok { return nil, nil, fmt.Errorf( - "cannot inject a default into column %q: the role's schema does not carry it", name) + "cannot inject a default into column %q: the table does not have it", name) } } + if len(wire) == 0 { + return nil, nil, errors.New("the role may write no plain or DEFAULT column of this table, so it has no row to publish") + } return cols, wire, nil } +// storedDefault is an expression for what ClickHouse stores in column c when an +// INSERT omits it: its own DEFAULT expression, or its type's default value +// (defaultValueOfTypeName, ClickHouse's own answer, measured to compile for +// every type family the 26.8 artifact accepts, Nullable and LowCardinality +// included). +func storedDefault(lib *chtypes.Library, c discovery.Column) (string, error) { + if c.DefaultKind == "DEFAULT" && c.DefaultExpression != "" { + return c.DefaultExpression, nil + } + typ, err := lib.QuoteLiteral(c.Type) + if err != nil { + return "", fmt.Errorf("cannot quote the type of column %q: %w", c.Name, err) + } + return "defaultValueOfTypeName(" + typ + ")", nil +} + type roleEntry struct { key string table *Table // nil: this shape does not compile diff --git a/internal/typelayer/roletable_test.go b/internal/typelayer/roletable_test.go index 6906b5f3..72066470 100644 --- a/internal/typelayer/roletable_test.go +++ b/internal/typelayer/roletable_test.go @@ -2,6 +2,7 @@ package typelayer import ( "encoding/json" + "errors" "runtime" "testing" @@ -52,15 +53,17 @@ func TestRoleTable_IdentityShapeIsTheBaseTable(t *testing.T) { assert.Equal(t, 0, base.roles.len()) } -// TestRoleTable_DeniedColumnIsAbsentFromTheSchema: a column the role may not -// write is simply not in the compiled DDL, so a record -// naming it is ClickHouse's own per-row code 117 rather than a Go key walk's -// 403 — and the exported row carries the ROLE's column list. -func TestRoleTable_DeniedColumnIsAbsentFromTheSchema(t *testing.T) { +// TestRoleTable_DeniedColumnIsUnnamableAndOffTheWire: a column the role may +// not write is declared MATERIALIZED, so a record naming it is ClickHouse's +// own per-row code 117 rather than a Go key walk's 403 — and the exported row +// carries the ROLE's column list. +func TestRoleTable_DeniedColumnIsUnnamableAndOffTheWire(t *testing.T) { eng := testEngine(t, ordersTable()) tbl := roleTableFor(t, eng, RoleShape{Columns: []string{"id", "tenant", "amount"}}) assert.Equal(t, []string{"id", "tenant", "amount"}, tbl.WireColumns) + assert.Equal(t, []string{"id", "tenant", "secret", "amount"}, compiledColumnNames(tbl.pool.first().schema), + "still declared, so an expression or check over it compiles") batch, err := tbl.Ingest(FormatJSONEachRow, []byte( `{"id":1,"tenant":"acme","amount":5}`+"\n"+ @@ -142,15 +145,118 @@ func TestRoleTable_LiteralEscaping(t *testing.T) { } // TestRoleTable_UnparseableLiteralFailsClosed: a literal the column's reader -// cannot read is a compile refusal (ClickHouse code 6). It must be Unavailable -// — a 503 — never a handle that silently drops the injection. +// cannot read is a compile refusal (ClickHouse code 6). It must be a +// RoleRefused — never a handle that silently drops the injection, and never an +// Unavailable, which a caller answers with a retry hint. func TestRoleTable_UnparseableLiteralFailsClosed(t *testing.T) { eng := testEngine(t, ordersTable()) _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Defaults: map[string]string{"amount": "abc"}}) require.Error(t, err) - assert.True(t, IsUnavailable(err)) - assert.Contains(t, err.Error(), "orders") + refused, ok := errors.AsType[*RoleRefused](err) + require.True(t, ok, "%T: %v", err, err) + assert.False(t, IsUnavailable(err)) + assert.Equal(t, "orders", refused.Table) + assert.Contains(t, refused.Cause, "code 6") +} + +// TestRoleTable_DeniedColumnAnExpressionReadsStillCompiles: dropping a denied +// column would leave every expression over it uncompilable and refuse the +// whole role. Re-declared MATERIALIZED with what the server stores when the +// worker omits it, the role compiles, and a DEFAULT column the role DOES write +// is computed from the same value the server would use — its own DEFAULT, or +// its type's default. +func TestRoleTable_DeniedColumnAnExpressionReadsStillCompiles(t *testing.T) { + ts := &discovery.TableSchema{ + Name: "visits", + Columns: []discovery.Column{ + {Name: "page", Type: "String", Position: 1}, + {Name: "ip", Type: "String", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "'0.0.0.0'", Position: 2}, + {Name: "ip_hash", Type: "UInt64", HasDefault: true, DefaultKind: "MATERIALIZED", DefaultExpression: "cityHash64(ip)", Position: 3}, + {Name: "ip_len", Type: "UInt64", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "length(ip)", Position: 4}, + {Name: "n", Type: "Nullable(UInt8)", IsNullable: true, Position: 5}, + {Name: "n_set", Type: "UInt8", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "isNotNull(n)", Position: 6}, + }, + } + eng := testEngine(t, ts) + tbl, err := eng.RoleTable(tenant.Default, "visits", RoleShape{Columns: []string{"page", "ip_len", "n_set"}}) + require.NoError(t, err) + defer tbl.Release() + + assert.Equal(t, []string{"page", "ip_len", "n_set"}, tbl.WireColumns) + batch, err := tbl.Ingest(FormatJSONEachRow, []byte( + `{"page":"/home"}`+"\n"+`{"page":"/a","ip":"1.2.3.4"}`+"\n"+`{"page":"/b","n":1}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 3) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, `["\/home", 7, 0]`, string(batch.Rows[0].Line), "length('0.0.0.0'), and n at its type default NULL") + assert.Equal(t, 117, batch.Rows[1].Code) + assert.Contains(t, batch.Rows[1].Message, "ip") + assert.Equal(t, 117, batch.Rows[2].Code) + assert.Contains(t, batch.Rows[2].Message, "n") + + // A check over a denied column tests what the server will store there. + check, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"page":"/home"}`+"\n"+`{"page":"/b"}`+"\n"), + Predicate{Column: "ip", Op: "=", Values: []string{"0.0.0.0"}}) + require.NoError(t, err) + assert.Equal(t, []string{"", ""}, checkReasons(t, check)) + check, err = tbl.Ingest(FormatJSONEachRow, []byte(`{"page":"/home"}`+"\n"), + Predicate{Column: "ip", Op: "=", Values: []string{"1.2.3.4"}}) + require.NoError(t, err) + assert.Equal(t, []string{ReasonFilter}, checkReasons(t, check)) +} + +// TestRoleTable_ShapeWithNothingToWriteIsRefused: a role that may write no +// plain or DEFAULT column would publish rows with no columns at all. +func TestRoleTable_ShapeWithNothingToWriteIsRefused(t *testing.T) { + ts := ordersTable() + ts.Columns = append(ts.Columns, discovery.Column{ + Name: "double", Type: "UInt64", HasDefault: true, + DefaultKind: "MATERIALIZED", DefaultExpression: "amount * 2", Position: 5, + }) + eng := testEngine(t, ts) + + for _, cols := range [][]string{{}, {"double"}, {"nosuch"}} { + _, err := eng.RoleTable(tenant.Default, "orders", RoleShape{Columns: cols}) + refused, ok := errors.AsType[*RoleRefused](err) + require.True(t, ok, "%v: %T %v", cols, err, err) + assert.Contains(t, refused.Cause, "no plain or DEFAULT column") + } +} + +// TestRoleTable_EphemeralFollowsTheRoleColumns: a record may supply an +// EPHEMERAL column only when the role may write it. Denied, it is still +// declared — the DEFAULT over it compiles and takes its own default — and a +// record naming it is refused. +func TestRoleTable_EphemeralFollowsTheRoleColumns(t *testing.T) { + ts := &discovery.TableSchema{ + Name: "eph", + Columns: []discovery.Column{ + {Name: "page", Type: "String", Position: 1}, + {Name: "secret", Type: "String", Position: 2}, + {Name: "ip", Type: "String", DefaultKind: "EPHEMERAL", HasDefault: true, Position: 3}, + {Name: "ip_len", Type: "UInt64", DefaultKind: "DEFAULT", DefaultExpression: "length(ip)", HasDefault: true, Position: 4}, + }, + } + eng := testEngine(t, ts) + body := []byte(`{"page":"/a","ip":"1.2.3.4"}` + "\n") + + allowed, err := eng.RoleTable(tenant.Default, "eph", RoleShape{Columns: []string{"page", "ip", "ip_len"}}) + require.NoError(t, err) + batch, err := allowed.Ingest(FormatJSONEachRow, body) + allowed.Release() + require.NoError(t, err) + require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, `["\/a", 7]`, string(batch.Rows[0].Line)) + + denied, err := eng.RoleTable(tenant.Default, "eph", RoleShape{Columns: []string{"page", "ip_len"}}) + require.NoError(t, err) + batch, err = denied.Ingest(FormatJSONEachRow, append(body, `{"page":"/b"}`+"\n"...)) + denied.Release() + require.NoError(t, err) + assert.Equal(t, 117, batch.Rows[0].Code) + require.True(t, batch.Rows[1].Accepted, batch.Rows[1].Message) + assert.Equal(t, `["\/b", 0]`, string(batch.Rows[1].Line)) } // TestRoleTable_ContradictoryShapeIsAnError: a default for a column the shape @@ -187,10 +293,10 @@ func TestRoleTable_DefaultIntoAComputedColumnIsRefused(t *testing.T) { assert.Contains(t, err.Error(), "MATERIALIZED") // A computed column is kept whatever the allow-list says: dropping it would - // change what the server computes. + // change what the server computes. So is a denied one, MATERIALIZED. tbl := roleTableFor(t, eng, RoleShape{Columns: []string{"id", "amount"}}) assert.Equal(t, []string{"id", "amount"}, tbl.WireColumns) - assert.Equal(t, []string{"id", "amount", "double"}, compiledColumnNames(tbl.pool.first().schema)) + assert.Equal(t, []string{"id", "tenant", "secret", "amount", "double"}, compiledColumnNames(tbl.pool.first().schema)) } // TestRoleTable_CachedPerShapeAndGeneration: the same shape must reuse the @@ -298,13 +404,13 @@ func TestRoleTable_PoolGrowsUnderContentionToALowerCap(t *testing.T) { p := tbl.pool require.Len(t, p.list(), 1, "a quiet shape holds one handle") - assert.Equal(t, int64(maxRolePoolSize), p.limit.Load(), "capped below the base table's %d", maxPoolSize) + assert.Equal(t, int64(4), p.limit.Load(), "min(GOMAXPROCS=8, 4): capped below the base table's 8") - held := make([]*schemaSlot, 0, maxRolePoolSize+1) - for range maxRolePoolSize + 1 { + held := make([]*schemaSlot, 0, 5) + for range 5 { held = append(held, p.acquire()) } - assert.Len(t, p.list(), maxRolePoolSize, "busy handles grow the pool to its cap and no further") + assert.Len(t, p.list(), 4, "busy handles grow the pool to its cap and no further") for _, s := range held { p.release(s) } diff --git a/internal/typelayer/typelayer.go b/internal/typelayer/typelayer.go index 7b55aa8f..92440f7b 100644 --- a/internal/typelayer/typelayer.go +++ b/internal/typelayer/typelayer.go @@ -18,6 +18,7 @@ package typelayer import ( "fmt" "log/slog" + "slices" "strings" "sync" @@ -319,12 +320,13 @@ func (s *tenantSet) bind(serverVersion, serverTZ string, tables []*discovery.Tab // Compile outside every lock: a handle costs milliseconds and Table() // readers are on the request path. type pending struct { - ts *discovery.TableSchema - sig string - pool *pool - cause string - wire []string - cols map[string]filterColumn + ts *discovery.TableSchema + sig string + pool *pool + cause string + wire []string + inputs []string + cols map[string]filterColumn } s.mu.RLock() @@ -349,10 +351,12 @@ func (s *tenantSet) bind(serverVersion, serverTZ string, tables []*discovery.Tab } } p := pending{ts: ts, sig: sig} - p.pool, p.cause = compile(lib, ts) + decl := discoveredColumns(ts.Columns) + p.pool, p.cause = compile(lib, decl) if p.cause == "" { schema := p.pool.first().schema - p.wire = deriveWireColumns(schema, wireColumns(ts)) + p.wire = deriveWireColumns(schema, WireColumnsOf(ts)) + p.inputs = inputColumns(lib, decl, p.wire, nil) if p.cols, p.cause = declaredColumns(lib, schema, ts.Columns); p.cause != "" { p.pool.close() p.pool = nil @@ -374,7 +378,7 @@ func (s *tenantSet) bind(serverVersion, serverTZ string, tables []*discovery.Tab // Nobody holds a pointer to a new table yet, so it is filled before // it is published rather than swapped. t = &Table{Name: p.ts.Name, tenant: s.id, Generation: 1, roles: newRoleCache(roleCacheSize)} - t.install(p.pool, p.cause, p.wire, p.cols, p.sig, lib, p.ts.Columns) + t.install(p.pool, p.cause, p.wire, p.inputs, p.cols, p.sig, lib, p.ts.Columns) added[p.ts.Name] = t continue } @@ -382,7 +386,7 @@ func (s *tenantSet) bind(serverVersion, serverTZ string, tables []*discovery.Tab old := t.detachLocked() t.Generation++ t.roles = newRoleCache(old.roles.capacity()) - t.install(p.pool, p.cause, p.wire, p.cols, p.sig, lib, p.ts.Columns) + t.install(p.pool, p.cause, p.wire, p.inputs, p.cols, p.sig, lib, p.ts.Columns) t.mu.Unlock() old.close() } @@ -441,6 +445,9 @@ type Table struct { tenant tenant.ID mu sync.RWMutex + // inputs is the INSERT column list a name-addressed body is parsed with + // (see inputColumns); nil parses with no list, which reads WireColumns. + inputs []string // pool holds the identically-compiled handles; nil means unavailable. pool *pool // cols maps every column the compiled schema declares, of every kind, to @@ -470,11 +477,11 @@ func (t *Table) answers(sig string) bool { // install sets a freshly compiled shape. The caller holds t.mu exclusively, // or is the only goroutine that can see t. -func (t *Table) install(p *pool, cause string, wire []string, cols map[string]filterColumn, sig string, +func (t *Table) install(p *pool, cause string, wire, inputs []string, cols map[string]filterColumn, sig string, lib *chtypes.Library, discovered []discovery.Column, ) { t.pool, t.cause = p, cause - t.WireColumns, t.cols, t.sig = wire, cols, sig + t.WireColumns, t.inputs, t.cols, t.sig = wire, inputs, cols, sig t.lib, t.discovered = lib, discovered } @@ -517,17 +524,7 @@ func (t *Table) close(cause string) { // compiled on contention. The engine and TTL clauses are deliberately not // declared: chtypes declines engines it cannot model, and neither affects the // insert verdicts or filter semantics this package asks for. -func compile(lib *chtypes.Library, ts *discovery.TableSchema) (*pool, string) { - cols := make([]chtypes.DiscoveredColumn, 0, len(ts.Columns)) - for _, c := range ts.Columns { - cols = append(cols, chtypes.DiscoveredColumn{ - Name: c.Name, - Type: c.Type, - DefaultKind: c.DefaultKind, - DefaultExpression: c.DefaultExpression, - Position: c.Position, - }) - } +func compile(lib *chtypes.Library, cols []chtypes.DiscoveredColumn) (*pool, string) { ddl, err := lib.ReconstructDDL(cols) if err != nil { return nil, "cannot reconstruct column declarations: " + err.Error() @@ -535,6 +532,26 @@ func compile(lib *chtypes.Library, ts *discovery.TableSchema) (*pool, string) { return newPool(lib, ddl, poolSize()) } +// discoveredColumns is a table's columns as chtypes' DDL reconstruction takes +// them. +func discoveredColumns(src []discovery.Column) []chtypes.DiscoveredColumn { + cols := make([]chtypes.DiscoveredColumn, 0, len(src)) + for _, c := range src { + cols = append(cols, discoveredColumn(c)) + } + return cols +} + +func discoveredColumn(c discovery.Column) chtypes.DiscoveredColumn { + return chtypes.DiscoveredColumn{ + Name: c.Name, + Type: c.Type, + DefaultKind: c.DefaultKind, + DefaultExpression: c.DefaultExpression, + Position: c.Position, + } +} + // signature is the column shape a handle was compiled from. An unchanged // signature keeps the handle and the generation, so a refresh that discovers // nothing new costs no compiles and invalidates no cached filter. @@ -579,8 +596,12 @@ func deriveWireColumns(schema *chtypes.LoadedSchema, fallback []string) []string return out } -// wireColumns is deriveWireColumns' discovery-side fallback. -func wireColumns(ts *discovery.TableSchema) []string { +// WireColumnsOf is the wire column list of a discovered table — what a +// full-width envelope's Columns carry — computed from discovery alone, for a +// caller that holds no compiled handle (the stream's connect-time schema +// frame). It is deriveWireColumns' fallback, and answered identically to the +// handle on every artifact measured. +func WireColumnsOf(ts *discovery.TableSchema) []string { out := make([]string, 0, len(ts.Columns)) for _, c := range ts.Columns { switch c.DefaultKind { @@ -592,6 +613,97 @@ func wireColumns(ts *discovery.TableSchema) []string { return out } +// inputColumns is the INSERT column list a name-addressed body (JSONEachRow, +// CSVWithNames, TSVWithNames) is parsed with: the wire columns plus every +// EPHEMERAL column a record may supply, in declaration order. nil when no +// EPHEMERAL column qualifies — the body is then parsed with no list, which +// reads exactly the wire columns. writable reports whether the role may +// supply a column; nil means every column. +// +// Listing an EPHEMERAL column is what lets a record supply it at all. With no +// list, ClickHouse's readers know only the stored columns and refuse it as an +// unknown field (117). Listed, its value is read and is in scope for the +// DEFAULT expressions over it, which chtypes computes and exports, while the +// value itself is never exported (measured on the 26.8 artifact). Positional +// CSV and TSV never get the list: their fields map to the wire columns by +// position, and an extra slot would shift every field after it. +// +// An EPHEMERAL column qualifies only when a DEFAULT expression reads it and +// no expression the server computes does: +// +// - The worker inserts the exported row, so ClickHouse computes each +// MATERIALIZED column itself, with every EPHEMERAL column at its own +// default. A value one of those reads would be silently ignored, so the +// column stays unlisted and a record naming it is refused instead. +// - A DEFAULT is the only place an ephemeral value can go. And listing one +// no DEFAULT reads makes the 26.8 artifact reject, with no code, every +// record that omits it whenever the table computes any column (measured): +// an unread column would cost every record its verdict to accept a value +// that changes nothing. +func inputColumns(lib *chtypes.Library, cols []chtypes.DiscoveredColumn, wire []string, writable func(string) bool) []string { + var ephemeral []string + for _, c := range cols { + if c.DefaultKind == "EPHEMERAL" && (writable == nil || writable(c.Name)) && + readBy(lib, cols, c.Name, "DEFAULT") && !readBy(lib, cols, c.Name, "MATERIALIZED", "ALIAS", "EPHEMERAL") { + ephemeral = append(ephemeral, c.Name) + } + } + if len(ephemeral) == 0 { + return nil + } + listed := make(map[string]bool, len(wire)+len(ephemeral)) + for _, n := range wire { + listed[n] = true + } + for _, n := range ephemeral { + listed[n] = true + } + out := make([]string, 0, len(listed)) + for _, c := range cols { + if listed[c.Name] { + out = append(out, c.Name) + } + } + return out +} + +// readBy reports whether an expression of one of kinds reads column name, +// asking the compiler rather than parsing SQL in Go: the declaration list is +// compiled without name, every other kind's expression dropped, so a +// reference to name fails the compile. Any refusal counts as a read. It +// compiles only when a column of those kinds has an expression at all. +func readBy(lib *chtypes.Library, cols []chtypes.DiscoveredColumn, name string, kinds ...string) bool { + probe := make([]chtypes.DiscoveredColumn, 0, len(cols)) + reads := false + for _, c := range cols { + switch { + case c.Name == name: + continue + case c.DefaultExpression == "": + case slices.Contains(kinds, c.DefaultKind): + reads = true + case c.DefaultKind == "EPHEMERAL": + c.DefaultExpression = "" // still EPHEMERAL, at its type's default + default: + c.DefaultKind, c.DefaultExpression = "", "" + } + probe = append(probe, c) + } + if !reads { + return false + } + ddl, err := lib.ReconstructDDL(probe) + if err != nil { + return true + } + s, err := lib.CompileDDL(ddl, chtypes.WithCompileSettings(compileSettings)) + if err != nil { + return true + } + s.Close() + return false +} + // filterColumn is how render writes a predicate over one column. type filterColumn struct { ident string // lib.QuoteIdentifier's spelling @@ -603,10 +715,11 @@ type filterColumn struct { // backQuote, always quoted) and, for an integer column, the type its claims // are strictly cast to. It is the set render tests a predicate's column // against: answering "no such column" here keeps a misspelled policy from -// costing a compile and a log line per generation, and on a ROLE table it is -// what makes a filter over a denied column fail closed instead of compiling -// against a column that is not there. Quoting once per compile keeps a C call -// off render's per-event path. A non-empty second return is the cause. +// costing a compile and a log line per generation. On a ROLE table a column +// the role may not write is declared too (MATERIALIZED, see roleColumns), so a +// check over it tests the value the server will store. Quoting once per +// compile keeps a C call off render's per-event path. A non-empty second +// return is the cause. func declaredColumns(lib *chtypes.Library, schema *chtypes.LoadedSchema, fallback []discovery.Column) (map[string]filterColumn, string) { type named struct{ name, typ string } var cols []named From b6035d3c8fc29fc96b174665f696c4d509e6f213 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 08:09:45 -0400 Subject: [PATCH 38/70] fix(ingest): refuse content after a JSON array body A body declared application/json that opened with '[' had every depth-1 comma reframed as a record separator, including those after the array's closing ']': `[{"page":"a"},{"page":"b"}] {"page":"c"}` answered 200 and published three records, and a trailing object's own commas were split into separate records. Anything but whitespace after the closing ']' is now a whole-request 400, `invalid json: content after the closing ']' of the json array`, like an unbalanced array, and nothing is published. An array whose '[' is closed by a '}' is reported as unbalanced. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/api/ingest.go | 16 +++++------ internal/api/ingest_framing.go | 43 ++++++++++++++++++++++------- internal/api/ingest_framing_test.go | 31 +++++++++++++++------ internal/api/ingest_test.go | 28 +++++++++++++++++++ 4 files changed, 91 insertions(+), 27 deletions(-) diff --git a/internal/api/ingest.go b/internal/api/ingest.go index df97f0a9..708799c8 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -362,14 +362,14 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { batchShape := format.alwaysBatch() || first == '[' records := 1 if format == FormatJSON && first == '[' { - n, framed := reframeArray(body.Bytes()) - if !framed { - // Brackets that do not balance: a truncated upload or a structural - // syntax error. Neither can be salvaged per record, and reporting the - // records that did arrive as a complete batch is the failure this - // refusal exists to prevent. - slog.WarnContext(ctx, "ingest read error", "error", "unterminated json array", "table", table) - writeJSONError(w, http.StatusBadRequest, "invalid json: unterminated json array") + n, err := reframeArray(body.Bytes()) + if err != nil { + // Brackets that do not balance — a truncated upload or a structural + // syntax error — or a tail after the array. None can be salvaged per + // record, and reporting what did frame as a complete batch is the + // failure this refusal exists to prevent. + slog.WarnContext(ctx, "ingest read error", "error", err.Error(), "table", table) + writeJSONError(w, http.StatusBadRequest, "invalid json: "+err.Error()) return } records = n diff --git a/internal/api/ingest_framing.go b/internal/api/ingest_framing.go index d4ebc67e..338860c5 100644 --- a/internal/api/ingest_framing.go +++ b/internal/api/ingest_framing.go @@ -1,6 +1,9 @@ package api -import "encoding/json" +import ( + "encoding/json" + "errors" +) // Framing: everything ingest reads out of the bytes it handles, the request // body and the rows ClickHouse exported from it. It is deliberately small — @@ -40,15 +43,24 @@ import "encoding/json" // A raw newline inside a string is illegal JSON, so leaving those alone costs // nothing and keeps the caller's bytes the caller's. // -// ok is false when the brackets do not balance — a truncated upload, or a -// structural syntax error — which is a whole-request 400. Nothing is published -// from a body we cannot frame. -func reframeArray(b []byte) (elements int, ok bool) { +// An error is a whole-request 400, and nothing is published from a body we +// cannot frame: errUnterminatedArray when the brackets do not balance — a +// truncated upload, or a structural syntax error — and errAfterArray when +// anything but whitespace follows the array's closing ']'. That tail is not a +// record of the array, and framing it as more records would publish what the +// caller never put in the batch. +func reframeArray(b []byte) (elements int, err error) { depth, commas := 0, 0 - sawValue := false + sawValue, closed := false, false inStr, esc := false, false for i := range b { c := b[i] + if closed { + if c != ' ' && c != '\t' && c != '\n' && c != '\r' { + return 0, errAfterArray + } + continue + } switch { case esc: esc = false @@ -66,8 +78,12 @@ func reframeArray(b []byte) (elements int, ok bool) { } case c == ']' || c == '}': depth-- - if c == ']' && depth == 0 { + if depth == 0 { + if c != ']' { + return 0, errUnterminatedArray // the array's '[' closed by a '}' + } b[i] = ' ' + closed = true } case c == ',' && depth == 1: b[i] = '\n' @@ -80,14 +96,21 @@ func reframeArray(b []byte) (elements int, ok bool) { } } if depth != 0 || inStr { - return 0, false + return 0, errUnterminatedArray } if !sawValue { - return 0, true // `[]`, possibly with whitespace inside + return 0, nil // `[]`, possibly with whitespace inside } - return commas + 1, true + return commas + 1, nil } +// The two ways reframeArray refuses a body, each the tail of the caller's +// "invalid json: …" 400. +var ( + errUnterminatedArray = errors.New("unterminated json array") + errAfterArray = errors.New("content after the closing ']' of the json array") +) + // cellAt returns the k-th top-level cell of one JSONCompactEachRow line — a // `[v0, v1, …]` array as ClickHouse's own writer produced it — without decoding // the row. Leading and trailing whitespace around the cell is trimmed; the cell diff --git a/internal/api/ingest_framing_test.go b/internal/api/ingest_framing_test.go index c400d1bd..632f4ebe 100644 --- a/internal/api/ingest_framing_test.go +++ b/internal/api/ingest_framing_test.go @@ -20,6 +20,7 @@ func TestReframeArray(t *testing.T) { want string count int ok bool + err error // the refusal when !ok }{ { name: "compact array becomes one record per line", @@ -85,20 +86,32 @@ func TestReframeArray(t *testing.T) { // only to keep them separate so the objects around them survive. count: 3, ok: true, }, - {name: "a truncated array does not balance", body: `[{"a":1}`, ok: false}, - {name: "a trailing comma cut off does not balance", body: `[{"a":1},`, ok: false}, - {name: "a bare open bracket does not balance", body: `[`, ok: false}, - {name: "a cut-off element does not balance", body: `[{"a":1},{"b`, ok: false}, - {name: "a structural syntax error does not balance", body: `[{"a":1}, {bad]`, ok: false}, + { + name: "whitespace after the array is layout", + body: "[{\"a\":1}] \r\n\t", + want: " {\"a\":1} \r\n\t", + count: 1, ok: true, + }, + {name: "a truncated array does not balance", body: `[{"a":1}`, err: errUnterminatedArray}, + {name: "a trailing comma cut off does not balance", body: `[{"a":1},`, err: errUnterminatedArray}, + {name: "a bare open bracket does not balance", body: `[`, err: errUnterminatedArray}, + {name: "a cut-off element does not balance", body: `[{"a":1},{"b`, err: errUnterminatedArray}, + {name: "a structural syntax error does not balance", body: `[{"a":1}, {bad]`, err: errUnterminatedArray}, + {name: "an array closed by a brace does not balance", body: `[{"a":1}}`, err: errUnterminatedArray}, + {name: "an object after the array is not a record of it", body: `[{"page":"a"},{"page":"b"}] {"page":"c","x":1}`, err: errAfterArray}, + {name: "a second array after the first", body: `[{"a":1}][{"a":2}]`, err: errAfterArray}, + {name: "a stray closing bracket after the array", body: `[{"a":1}]]`, err: errAfterArray}, + {name: "any other byte after the array", body: "[{\"a\":1}]\nx", err: errAfterArray}, } { t.Run(tt.name, func(t *testing.T) { t.Parallel() b := []byte(tt.body) - count, ok := reframeArray(b) - require.Equal(t, tt.ok, ok, "balance") + count, err := reframeArray(b) if !tt.ok { + require.ErrorIs(t, err, tt.err) return } + require.NoError(t, err) assert.Equal(t, tt.count, count, "element count") assert.Equal(t, tt.want, string(b), "rewritten body") }) @@ -113,12 +126,12 @@ func TestReframeArray_DestroysWhatItMustNotSee(t *testing.T) { t.Parallel() obj := []byte(`{"a":1,"b":2}`) - reframeArray(obj) + _, _ = reframeArray(obj) // the damage, not the verdict, is what this pins assert.Equal(t, "{\"a\":1\n\"b\":2}", string(obj), "a bare object's own commas are at depth 1 — the handler must never send one here") ndjson := []byte("{\"a\":1,\"b\":2}\n{\"a\":3,\"b\":4}\n") - reframeArray(ndjson) + _, _ = reframeArray(ndjson) assert.NotContains(t, string(ndjson), `{"a":1,"b":2}`, "an NDJSON body is destroyed too — same reason, same gate") } diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index d500feed..6ff77141 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -2287,6 +2287,34 @@ func TestIngest_JSONArray_Truncated_Fatal(t *testing.T) { } } +// TestIngest_JSONArray_TrailingContent_Fatal: bytes after the array's closing +// ']' are not records of it. Framed as more records they would be published +// (a trailing object, its commas rewritten as separators); the whole request +// fails instead, like an unbalanced array, and nothing is published. Trailing +// whitespace is layout and still ingests. +func TestIngest_JSONArray_TrailingContent_Fatal(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + + for _, body := range []string{ + `[{"page":"a"},{"page":"b"}] {"page":"c"}`, + `[{"page":"a"}] {"page":"c","button":"x"}`, + `[{"page":"a"}][{"page":"b"}]`, + } { + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "application/json", body))) + assert.Equal(t, http.StatusBadRequest, w.Code, "%s: body=%s", body, w.Body.String()) + assert.Equal(t, "invalid json: content after the closing ']' of the json array", jsonErrorMessage(t, w), body) + } + assert.Empty(t, pub.Messages) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "application/json", "[{\"page\":\"a\"}]\n\n"))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.Len(t, pub.Messages, 1) +} + func TestIngest_JSONArray_Empty(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} From 3c8542a314d359da0b8a21804dfbe88a4c14e210 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 08:11:06 -0400 Subject: [PATCH 39/70] test(integration): stored rows for denied, stamped and ephemeral columns End-to-end cases for the role-shape fixes, through HTTP ingest, the broker and the worker's INSERT: a column denied to a role but read by a DEFAULT and a MATERIALIZED column stores the server's own values; an _eq check on a column the role may not write stamps the claim; an EPHEMERAL value (JSON and CSV with a header) lands as the DEFAULT computed from it. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- tests/integration/ingest_test.go | 79 ++++++++++++++++++++++++++++++-- 1 file changed, 74 insertions(+), 5 deletions(-) diff --git a/tests/integration/ingest_test.go b/tests/integration/ingest_test.go index 21ecf199..d1b3449a 100644 --- a/tests/integration/ingest_test.go +++ b/tests/integration/ingest_test.go @@ -311,11 +311,11 @@ func TestIngest_WithNamesUnknownHeader_Is400WithCode117(t *testing.T) { eventuallyRows(t, table, "1", 1) } -// A column the role may not write is not in the schema its records are -// parsed against, so a record naming one is ClickHouse's per-record refusal — -// 400 with code 117 — where it used to be the gateway's 403. The same role -// writing without it still lands, the column taking the table's default -// rather than any value of the caller's. +// A column the role may not write is one no INSERT may name in the schema its +// records are parsed against, so a record naming one is ClickHouse's +// per-record refusal — 400 with code 117 — where it used to be the gateway's +// 403. The same role writing without it still lands, the column taking the +// table's default rather than any value of the caller's. func TestIngest_DeniedColumn_IsClickHouseCode117(t *testing.T) { t.Parallel() table := createTable(t, "user_id String, secret String", "ORDER BY user_id") @@ -335,6 +335,75 @@ func TestIngest_DeniedColumn_IsClickHouseCode117(t *testing.T) { eventuallyRows(t, table, "user_id = 'd1'", 0) } +// Denying a column that DEFAULT and MATERIALIZED expressions read leaves the +// role able to insert, and what lands is what the server computes with the +// denied column at its own default: the DEFAULT column the role writes is +// computed at the gateway from that same default, so the two agree. +func TestIngest_DeniedColumnReadByExpressions_StoresTheServersValues(t *testing.T) { + t.Parallel() + table := createTable(t, + "user_id String, ip String DEFAULT '0.0.0.0', ip_len UInt64 DEFAULT length(ip), "+ + "ip_hash UInt64 MATERIALIZED cityHash64(ip)", + "ORDER BY user_id") + withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{ + table: {"writer": {Insert: &policy.InsertPermissions{DenyColumns: []string{"ip"}}}}, + }}) + writer := bearer(t, "writer", nil) + + status, body := postIngest(t, table, "application/json", `{"user_id":"m1"}`, writer) + require.Equal(t, http.StatusOK, status, "body=%v", body) + status, body = postIngest(t, table, "application/json", `{"user_id":"m2","ip":"1.2.3.4"}`, writer) + require.Equal(t, http.StatusBadRequest, status, "body=%v", body) + assert.EqualValues(t, 117, body["exception_code"]) + + eventuallyRows(t, table, "user_id = 'm1' AND ip = '0.0.0.0' AND ip_len = 7 AND ip_hash = cityHash64('0.0.0.0')", 1) + eventuallyRows(t, table, "user_id = 'm2'", 0) +} + +// An _eq check on a column the role may not otherwise write stamps it: a +// record omitting it lands with the claim, one supplying the claim lands, and +// one supplying anything else is refused. +func TestIngest_AutoInject_StampsAColumnTheRoleMayNotWrite(t *testing.T) { + t.Parallel() + table := createTable(t, "user_id String, org_id String", "ORDER BY user_id") + tmpl := "{{ jwt.org_id }}" + withPolicy(t, policy.Policy{Tables: map[string]policy.TablePolicy{ + table: {"writer": {Insert: &policy.InsertPermissions{ + AllowColumns: []string{"user_id"}, + Check: map[string]policy.Filter{"org_id": {Eq: &tmpl}}, + }}}, + }}) + writer := bearer(t, "writer", map[string]any{"org_id": "acme"}) + + status, body := postIngest(t, table, "application/json", `{"user_id":"s1"}`, writer) + require.Equal(t, http.StatusOK, status, "absent → stamped; body=%v", body) + status, body = postIngest(t, table, "application/json", `{"user_id":"s2","org_id":"acme"}`, writer) + require.Equal(t, http.StatusOK, status, "supplied and matching; body=%v", body) + status, body = postIngest(t, table, "application/json", `{"user_id":"s3","org_id":"other"}`, writer) + require.Equal(t, http.StatusForbidden, status, "supplied and not matching; body=%v", body) + + eventuallyRows(t, table, "user_id IN ('s1', 's2') AND org_id = 'acme'", 2) + eventuallyRows(t, table, "user_id = 's3'", 0) +} + +// An EPHEMERAL column's value feeds the DEFAULT over it, in the formats that +// name their columns, and the stored row holds what that DEFAULT computed. +func TestIngest_EphemeralColumn_FeedsTheStoredDefault(t *testing.T) { + t.Parallel() + table := createTable(t, "user_id String, raw String EPHEMERAL '', raw_len UInt64 DEFAULT length(raw)", "ORDER BY user_id") + + status, body := postIngest(t, table, "application/json", `{"user_id":"e1","raw":"abcd"}`, "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + status, body = postIngest(t, table, "text/csv; header=present", "raw,user_id\nabcdef,e2\n", "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + status, body = postIngest(t, table, "application/json", `{"user_id":"e3"}`, "") + require.Equal(t, http.StatusOK, status, "body=%v", body) + + eventuallyRows(t, table, "user_id = 'e1' AND raw_len = 4", 1) + eventuallyRows(t, table, "user_id = 'e2' AND raw_len = 6", 1) + eventuallyRows(t, table, "user_id = 'e3' AND raw_len = 0", 1) +} + // An _eq check fills a record that omits its column from the claim, keeps a // record's own value when it matches, and refuses one that does not (403, // nothing stored). From 330fd09689a0731e32593d6662720ced8bb457ea Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 08:16:38 -0400 Subject: [PATCH 40/70] test(typelayer): use the in-package engine helper Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- internal/typelayer/ingest_test.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/internal/typelayer/ingest_test.go b/internal/typelayer/ingest_test.go index eb69bf20..34fefbbf 100644 --- a/internal/typelayer/ingest_test.go +++ b/internal/typelayer/ingest_test.go @@ -266,7 +266,7 @@ func TestInsertSettings_BestEffortDateTime(t *testing.T) { // in UTC at the column's scale, whatever the input spelling or the column's // zone, as the query paths render it; Date and Date32 keep their own form. func TestIngest_DateTimeExportsAsRFC3339UTC(t *testing.T) { - eng := TestEngine(t, &discovery.TableSchema{Name: "times", Columns: []discovery.Column{ + eng := testEngine(t, &discovery.TableSchema{Name: "times", Columns: []discovery.Column{ {Name: "dt", Type: "DateTime", Position: 1}, {Name: "dt3", Type: "DateTime64(3)", Position: 2}, {Name: "dt6", Type: "DateTime64(6)", Position: 3}, From 3674b796f8ad1ae9751bd17b48ac0a842cd2c5d5 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 08:19:53 -0400 Subject: [PATCH 41/70] docs: match the role-shape, EPHEMERAL, timestamp and read-cap fixes Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- AGENTS.md | 2 +- CHANGELOG.md | 13 +++++++++---- docs/src/content/docs/access-control.mdx | 4 ++-- docs/src/content/docs/api.md | 12 ++++++++---- docs/src/content/docs/architecture.md | 15 ++++++++------- docs/src/content/docs/configuration.mdx | 2 +- docs/src/content/docs/deployment.md | 2 +- docs/src/content/docs/development.md | 2 +- docs/src/content/docs/sdk/reference.md | 2 +- docs/src/content/docs/sdk/streaming.md | 6 ++++-- docs/src/content/docs/settings-directory.mdx | 2 +- 11 files changed, 37 insertions(+), 25 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index a1091be6..74126f00 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -47,7 +47,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes - **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row via `typelayer`, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns, plus a `DEFAULT ''` per `_eq` check column — which is how column policy and auto-inject are answered with no Go-side record inspection. `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) +- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns (a denied column stays as `MATERIALIZED` of its default, so naming it is code 117), plus a `DEFAULT ''` per `_eq` check column, and any `EPHEMERAL` column the role may write that a `DEFAULT` reads — which is how column policy and auto-inject are answered with no Go-side record inspection; a role shape's handle pool grows to `min(GOMAXPROCS, 4)` under load. A role schema that cannot compile is `500` with `retryable:false`. Test helpers live in `internal/typelayer/typelayertest` (`TestEngine`, `SkipWithoutArtifact`). `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) - **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions diff --git a/CHANGELOG.md b/CHANGELOG.md index cf08e350..f1af194d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -63,9 +63,9 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `builds:` and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes --notes-start-tag "$(git describe --match 'v*')"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. -- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date`, `Date32` and `Time` are unaffected) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. -- **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, without the columns it may not write, instead of walking a decoded record's keys. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a checked column behaves exactly like omitting it. An `_eq` check on a column the role may not write is still injected and published. A role whose column policy cannot be compiled is refused with a non-retryable error and the cause is logged, rather than answered with a `503` that a retry could not fix. +- **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, instead of walking a decoded record's keys. A column the role may not write stays in that schema as `MATERIALIZED` of its default, so naming it is refused while expressions that read it keep working and the stored row holds the table's default. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a checked column behaves exactly like omitting it. An `_eq` check also covers a column the role may not otherwise write: a record that omits it is filled with the required value and published, one that supplies exactly that value is accepted (**0.1.0 answered `403 column "x" not allowed for insert` to the correct value**), and any other value is `403 check failed for column "x"`. A role whose schema cannot be compiled this way, or that may write no column of the table, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After` (the cause is logged once a minute), rather than a `503` that a retry could not fix; a policy `check` on an `EPHEMERAL` column is still `403`. - **WaveHouse now requires cgo, and supported platforms narrow to darwin/arm64, linux/amd64, linux/arm64** (BREAKING; `go.mod`, `scripts/build.sh`, `.goreleaser.yaml`, `deployments/Dockerfile`, `deployments/Dockerfile.goreleaser`, `Makefile`, `internal/config/config.go`, `config.yaml`, `chtypes.lock` (new), `scripts/fetch-chtypes.sh` (new), `.github/actions/setup-env/action.yml`, `.github/workflows/ci.yml`): the native type layer above needs cgo for `dlfcn` (no C library linked, no header). cgo is now unconditional: `CGO_ENABLED=0` is gone from every build path, and the `make audit-cgo` target that policed the old no-cgo build has been **removed** along with it (`make binary-analysis` is now `size` + `deadcode`). Because chtypes publishes artifacts only for darwin-arm64, linux-amd64 and linux-arm64, **Windows, FreeBSD and darwin/amd64 builds are discontinued** — `.goreleaser.yaml`'s matrix drops from 8 targets to 3, and the release archives/checksums/GHCR image narrow to match. The runtime image moves from an Alpine/musl builder + `distroless/static` to `golang:1.27-bookworm` (glibc, ships gcc) + `distroless/cc-debian12` (glibc + libstdc++, which the SDK's shared library needs) and bakes the pinned chtypes artifact into the image at `/opt/chtypes/artifacts` via a new `chtypes.lock` (exact file + sha256 per platform/line) and `scripts/fetch-chtypes.sh --frozen` wrapper, so the container has no first-request download. `go.mod` moves to `go 1.27`. Only processes running the `api` role load the artifact, so an ingest-worker-only or sweeper-only process needs none installed; the glibc requirement is 2.34 or later. New boot config: `clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY` lets an operator point at an explicit registry directory instead of the SDK's own search path (the shipped image instead sets the SDK's own `CHTYPES_REGISTRY` env var directly). CI's `unit`/`integration`/`e2e` jobs fetch and cache the pinned artifact (`setup-env`'s new `chtypes` input) and set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` so a missing artifact fails the job instead of silently skipping the chtypes-backed tests. `GOLANGCI_LINT_VERSION` bumped `v2.11.4` → `v2.13.2`: the `go 1.27` bump panics `v2.11.4`'s type checker on every package; `v2.13.0` is the oldest release whose changelog claims go1.27 support, but it panics in this tree for a different reason (`nilness`/`honnef.co/go/tools@v0.8.0-rc.1` crashing while analyzing a third-party dependency), fixed once that dependency moves past its release candidate in `v2.13.1`. The cross-toolchain approach this bullet originally described was replaced before landing — see the release-pipeline entry above. @@ -79,9 +79,14 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Ingest now requires a declared `Content-Type`, and it is authoritative** (BREAKING; `internal/api/record_reader.go`, `internal/api/ingest.go`, `clients/ts/src/table.ts`, `docs/src/content/docs/{api.md,architecture.md,sdk/queries.md}`): `POST /v1/ingest` used to sniff the body and treat the header as a hint — the first non-whitespace byte chose between a single object and an array, and an `application/x-ndjson` body that happened to start with `[` was silently re-read as a JSON array. A request that declares **no** `Content-Type`, or one whose media type is not in the accepted list, is now rejected with `415` before the body is parsed, naming every accepted type (`application/json`, `application/x-ndjson`, `application/ndjson`, `application/jsonl`, `application/jsonlines`) and quoting what was declared, bounded to four distinct header lines each capped at 128 bytes. The header is parsed per RFC 9110 §8.3 via `mime.ParseMediaType` rather than by hand, so the grammar's rules apply — parameters never affect the format (with the one exception below), and a comma inside a quoted value is data. Because `Content-Type` is a **singleton** field (§5.3 forbids repeating it), anything that is not exactly one readable media type is refused; the one accommodation is that repeated header lines are all resolved and accepted when they agree, since honoring just the first would let an NDJSON body be read as one JSON object and drop every record past it. A comma-joined value gets no such accommodation — §8.3 warns that taking a member of the pseudo-list is itself an interoperability and security hazard. **Four additional shapes 0.1.0 accepted now `415`** (beyond repeated header lines that disagree, which it also accepted): a present-but-empty header (a bare `Content-Type:` line, or one that is only whitespace); a value with a trailing or leading comma (`application/json,`); a comma-joined value that does not parse as a single media type (`application/json, application/json` — but a comma *inside a quoted parameter value* is legal data, so `application/json; a=", application/x-ndjson; b="` is one media type and is accepted); and a malformed parameter on a line that *also* carries a comma, which is refused rather than guessed at because the comma may be a second declaration joined on — so `application/json; profile="a,b"; charset` is a `415` while `application/json; profile="a,b"` and `application/json; charset` are each accepted ([#563](https://github.com/Wave-RF/WaveHouse/issues/563)). A repeated parameter name is *not* among them: the media type is re-parsed alone, so `; charset=a; charset=b` reads as `application/json` like every other malformed parameter. The declaration now decides the format outright: a body declared as NDJSON is read as NDJSON whatever its first byte, so a line that isn't a JSON object fails as a **per-record** error through the existing batch-result path instead of re-framing the whole request. The body still picks arity *within* the JSON family — `[` is an array, anything else a single object — because those are the same format at different lengths. Clients that relied on the sniffer must now send a header; the TS SDK already sent one on both paths (`application/json` for a single object, `application/x-ndjson` for arrays and `insertNDJSON`) and now states it at each call site rather than leaning on the request default, and every `curl` example in the docs already carried one. The format is modeled as an `IngestFormat` where the sniffing used to live, with the slot for CSV kept where the old comment marked it. +- **An `EPHEMERAL` column is accepted as ingest input only where it can be honoured** (`internal/typelayer/{typelayer,ingest}.go`, `internal/api/ingest.go`, `docs/src/content/docs/{api.md,sdk/reference.md,deployment.md,architecture.md}`): a record in a JSON body, or in CSV/TSV with `header=present`, may name an `EPHEMERAL` column when the role may write it, a `DEFAULT` column reads it and no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` expression does; its value feeds that `DEFAULT` and is never stored, selected or published. Any other `EPHEMERAL` column is refused like an unknown one (`400`, code 117), because ClickHouse computes `MATERIALIZED` columns from the published row, which never carries the ephemeral value. Positional CSV/TSV (no header, or `header=absent`) carry the wire columns only. A policy `check` on an `EPHEMERAL` column is still `403`. +- **A JSON array body with content after its closing `]` is refused** (`internal/api/ingest_framing.go`): the request answers `400 {"error":"invalid json: content after the closing ']' of the json array"}` and publishes nothing. +- **A role's ingest handle pool grows under load** (`internal/typelayer/pool.go`): a role shape's pool of compiled handles starts at one and grows to `min(GOMAXPROCS, 4)` when every handle is busy, as a base table's grows to `min(GOMAXPROCS, 8)`, so concurrent inserts by one role no longer serialize on a single handle. +- **Client-side timestamp comparison in the SDK is by instant** (`clients/ts/src/timestamp.ts` (new), `clients/ts/src/{query-builder,stream/live-query}.ts`): two strings that both read as timestamps are compared to the nanosecond for `=`, `!=`, `in`, `>`, `>=`, `<`, `<=` in stream `where` filters, with a zone-less value read as UTC; the live-query backfill seam compares at the coarser of the two precisions, so an event in the boundary row's own millisecond is covered. Other strings keep strict equality and string order. +- **Test helpers moved to `internal/typelayer/typelayertest`** (internal): `TestEngine`, `SkipWithoutArtifact` and `RequireEnv` no longer live in `internal/typelayer`, which now never imports `testing`. - **A policy `check` on a column the table cannot accept is now refused instead of silently unenforced** (BREAKING; `internal/api/ingest.go`, `internal/discovery/discovery.go`, `docs/src/content/docs/{api.md,access-control.mdx}`): a `check` clause naming a column the table does not have, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one can never be enforced — the published row carries one slot per insertable column, so an auto-injected value for anything outside that set is dropped on the way out, and an ephemeral column is never stored even though the row does carry it. The record inserted **without** the value the policy required and answered `200 {"ok":true}`. Demonstrated on this branch: a `check` of `tenant _eq {{ jwt.tenant }}` against a `MATERIALIZED tenant` column published `columns:["page"], row:["/a"]` — the tenant constraint absent from the row entirely. It is now refused, naming every offending column and the reason, on **every** insert by that role until the policy or the table is corrected: a single-object request answers `403`, while a batch answers `200` with the same message against each record in `results` — the batch is still read to the end and reports per record, as it does for any other rejection. Policy validation cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**; `wavehouse validate` will not tell you. -- **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` / `InsertableColumns` / `InsertableColumnNames`, memoized per table since the ingest path would otherwise rebuild them once per record. `EPHEMERAL` columns are likewise left out of the envelope, and this entry's earlier claim that they stay insertable is superseded by the type-layer entry above: an `EPHEMERAL` value is accepted as input wherever the format names columns (the JSON family and the `…WithNames` formats) and feeds the `DEFAULT`s that read it, but it is never stored, selected or published, and a positional CSV or TSV body carries the wire columns only. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. +- **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` (the `InsertableColumns` / `InsertableColumnNames` lists it first added are removed again, an internal API: which columns a record may supply is the type layer's `Table.WireColumns`). `EPHEMERAL` columns are likewise left out of the envelope, and this entry's earlier claim that they stay insertable is superseded by the `EPHEMERAL` entry above. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. - **Ingest reads the request body up front, and the per-record decisions sit behind interfaces** (`internal/api/{ingest,ingest_seams,bufpool,record_reader}.go`, `internal/stream/hub.go`, `internal/ingest/compact.go`): responses are unchanged except at the body cap and one new read-failure body (`400 {"error":"invalid request body"}`, when the body cannot be read at all — a malformed transfer encoding or a truncated upload, which previously surfaced through the decoder as `invalid json`), and at the cap the `413` is now decided before any record is processed: an over-cap batch no longer ingests the prefix it had already decoded, and an over-cap single-object body whose first object was followed by an oversized tail — which used to answer `200` after ingesting that one object — now answers `413`. Both are improvements, since a client retrying a `413` can no longer double-insert a prefix, but they are behavior changes and the memory profile changes too (see below); this is the seam work the native type layer lands against. The handler now reads the whole (already `MaxBytesReader`-capped) body into a pooled `*bytes.Buffer` and runs the record readers over those bytes rather than the live connection — so the `413` surfaces at that read instead of mid-iteration (same status, same message), and the `415` is decided from the header before a single byte is read. Three decision points became interfaces with default implementations that delegate to today's code unchanged: `RecordValidator` (schema validation + timestamp canonicalization — the two calls stay where they are, with the check-clause block between them, since merging them would move checks onto canonicalized values), `InsertChecker` (the `_eq` and `_in` comparisons), and `stream.RowEvaluator` (row visibility, reached by both the live fan-out and replay through the one shared admission step). **The memory profile is not unchanged, and that is the deliberate part.** Streaming meant peak resident bytes on the order of one record: the NDJSON path scanned line by line and the array path let `json.Decoder` compact after each element. Peak is now O(body) per in-flight request — and `bytes.Buffer` grows by doubling, so the peak allocation can exceed the body cap before `MaxBytesReader` errors. `maxPooledBufferBytes` (1 MiB) caps what a request hands *back* to the pool, not its peak, and nothing in `internal/api` bounds total in-flight bytes, so the ceiling is concurrency × the 16 MiB data-plane cap — which has no operator knob (`maxRequestBytes` is test-only), so the outer limit is the reverse proxy's, which the reverse-proxy guide already advises setting. Kept because it is the shape the native type layer lands against, which needs the body addressable rather than consumed; a bound on total in-flight ingest bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). Operators fronting large batches at high concurrency should size for it or cap body size at the proxy. All three are nil-safe: an un-wired handler or `Hub` uses the default rather than panicking past the check. Also new: `ingest.EncodeCompactRow`, which renders a record as one `JSONCompactEachRow` line — inert in this commit, and the encoder every published row went through at that point in the release (superseded: the published row is now ClickHouse's own export, and `EncodeCompactRow`, `RecordValidator` and `InsertChecker` are gone — see the type-layer entry). @@ -93,7 +98,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The landing page's live demo now reads from the stats deployment's new WaveHouse Cloud backend** (`docs/src/components/LiveDemo.astro`, `docs/scripts/screenshot.mjs`): the GitHub-activity dogfood deployment behind the hero panel (Wave-RF/WaveHouse-Stats) moved off its self-managed AWS infrastructure onto WaveHouse Cloud, so `BASE_URL` — the origin `@wavehouse/sdk` queries in the visitor's browser — points at `https://iefrrvavd5akvphk7pq3.wavehouse.app` instead of `https://stats.wavehouse.dev`, ahead of the AWS stack being torn down. The `PUBLIC_WAVEHOUSE_STATS_URL` build-time override is unchanged, so a fork or staging docs build still redirects the panel without a code edit. **`DEMO_HOST` deliberately stays `stats.wavehouse.dev`** — the demo *site* is still served there and is still what the panel's chrome label and "Full demo" link should show; the migration splits the site from the API origin behind it, and the two constants now carry comments saying so. Verified against the new deployment before the switch: all five pipes the panel reads (`gh_summary`, `gh_activity_recent`, `gh_events_per_minute`, and the pre-#19 `gh_stars_total` / `gh_forks_total` fallbacks) return `200` with the same row shapes, the structured-query backfill fallback (`POST /v1/query?table=gh_events`) matches its old-backend response byte for byte, `GET /v1/stream?table=gh_events` opens an SSE stream, and CORS is unchanged (`Access-Control-Allow-Origin: *`, `X-Cache` exposed) so the cross-origin browser reads keep working from the docs site. The new backend is already the live ingest target — it reported more recent events than the old one at cutover (5,175 vs 5,038 over 7d) — which is the other half of why the panel had to follow it. `screenshot.mjs`'s `networkidle` note is retargeted to "the stats demo backend" rather than naming a host it no longer connects to. -- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. Each tenant's reader connections are capped at its `max_open_conns`, per pool identity (URL, user, database and TLS settings). Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. +- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. The reader's HTTP connections are capped per connection tuple (URL, user, password, database, TLS) at the largest `max_open_conns` among the tenants sharing it. The 64 MiB cap is new on these two paths, so a pipe (which has no row limit) returning more than 64 MiB now fails; `/v1/query` normally stays under it through its default row cap. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. - **WH001 (no hard-wrapped prose) now applies to every tracked Markdown file, with no carve-out** (`.github/.markdownlint.json` (deleted), `.claude/.markdownlint.json` (deleted), `.claude/skills/integration-astro-view-transitions/` (deleted), `.markdownlint-cli2.jsonc`, `.github/workflows/README.md`, `.claude/skills/pm-triage/references/routine.md`, `AGENTS.md`, `scripts/docs-prose.sh`, `.github/prompts/docs-review.md`, `docs/src/content/docs/claude-code.md`, `docs/src/content/docs/development.md`, `.claude/agents/docs-reviewer.md`): two path-scoped configs had switched WH001 off under `.github/` and `.claude/` ever since [#489](https://github.com/Wave-RF/WaveHouse/pull/489) introduced the rule — baked in from the start rather than added in response to a discovered problem — which left the repo documenting the rule three ways and disagreeing with itself: `CONTRIBUTING.md` promises contributors `make lint` enforces it *everywhere*, while `AGENTS.md` and the `.markdownlint-cli2.jsonc` header wrote up the carve-out. Not theoretical: on [#520](https://github.com/Wave-RF/WaveHouse/pull/520) a reviewer correctly flagged a hard-wrapped bullet in `.github/workflows/README.md`, an agent pointed at `"WH001": false` for that path and pushed back, and the reviewer recorded a *learning* never to flag WH001 there — the wrong invariant, learned off the wrong side of the contradiction ([#521](https://github.com/Wave-RF/WaveHouse/issues/521)). Both configs are deleted — each held nothing but the override, so the root `.markdownlint.json` governs again — and the 51 hard-wrapped paragraphs they were hiding are joined: 41 in `.github/workflows/README.md` and 10 in `.claude/skills/pm-triage/references/routine.md`, mechanical joins with no wording changed and every fenced block, table row, and heading byte-identical either side of the reflow. Deleted with them: the wizard-installed PostHog skill at `.claude/skills/integration-astro-view-transitions/` — 9 files, ~1,456 lines, including an 809-line `EXAMPLE.md` copied wholesale from `PostHog/context-mill`. Its integration job finished in [#277](https://github.com/Wave-RF/WaveHouse/pull/277), nothing in the repo calls it, and the docs-site setup it once described is documented where it belongs — in `docs/src/components/PostHog.astro` and this file. Keeping unowned third-party prose in the tree means content that drifts silently on every upstream bump and that nobody here reviews; it was also the single file that would have needed a special-case lint exclusion, so removing it is what lets WH001 apply with **no exception at all** rather than one documented one. Its two inventory rows in `claude-code.md` go with it, as does the now-dead `docs/posthog-setup-report.md` entry in the `scripts/docs-prose.sh` denylist (the wizard's other artifact, deleted back in [#502](https://github.com/Wave-RF/WaveHouse/pull/502)) and the copies of that denylist in `AGENTS.md` and `.github/prompts/docs-review.md`, which the script's header requires be kept in lockstep. Review of the change then turned up four more things the exclusion had been hiding, all fixed here: **WH001 has a blind spot** — `no-hard-wrapped-prose.mjs` classifies any line indented four or more spaces as an indented code block, so a *nested* list item is never joined, which left three hard-wrapped bullets in `.github/workflows/README.md` §"Adding a job" that the autofix could not see (unwrapped by hand; they were the last hard-wrapped prose paragraphs in the repo) and made `AGENTS.md`'s and `development.md`'s "a list item is joined as a unit" wrong for nested items (both now state the four-space caveat); the `scripts/docs-prose.sh` header told readers to keep its denylist in lockstep with **two** sibling copies when there are **three** — the missed one being `.claude/agents/docs-reviewer.md`, the gating subagent's own system prompt, which had in fact been silently out of sync for the whole life of the `posthog-setup-report.md` exclusion; the `.markdownlint-cli2.jsonc` header's "applies to every tracked Markdown file" was exact for WH001 but not WH002, which returns early on anything that isn't `.mdx`; and the job-graph diagram omitted `docs-deploy`'s `needs` edges from `unit`, `integration`, and `e2e`, contradicting invariant 2 three lines below it. The denylist also drops its `PERF-CLAIMS-REVIEW.md` entry — unlike the wizard artifact this one names a file that was **never tracked** at all, so it guarded a hypothetical; the list's other general cases are patterns (`*.draft.md`, `*.old.md`) that already cover a one-off review document, and a literal filename restated in four places is the outlier. `scripts/docs-prose.sh all` still resolves the same 27-file prose set. diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 4e401791..4aefaf3c 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -192,7 +192,7 @@ The rules, in order: On a structured query (`POST /v1/query?table={table}`) the allowlist is a **hard cap on every column the query references — in any clause**: the projection, an aggregation argument, `filters`, `group_by`, `order_by`, and `time_range`. Naming a disallowed column anywhere is rejected with `403 column "x" not allowed`. A full-row read is requested explicitly with `"select_all": true`, which expands to exactly the columns the role may read — never a raw `SELECT *` that could include a denied column; if the role is allowed *no* columns, the read is rejected (`403`) rather than returning empty rows. **Omitting `columns` (or sending `[]` / `""`) returns nothing** — a request for no data — so a hidden column can't leak by being left out, grouped on, or filtered on to infer its values. (Note: in a query, `["*"]` is the *literal column named `*`*, not a wildcard — use `select_all` for all columns. In `allow_columns`, `["*"]` is still the all-columns wildcard.) On live streams, denied columns are silently **stripped** from each event rather than rejecting the connection. The structured-query and live-stream paths defer to the **same** per-column decision (`IsColumnAllowed`), so the two read surfaces enforce identical column visibility and can't drift apart. -On ingest (`POST /v1/ingest?table={table}`) the same lists cap what a role may write, and a column it may not write is indistinguishable from one the table does not have: the record is refused by ClickHouse's own parser with code `117` (`Unknown field found while parsing …`), a `400` — per record in a batch, or for the whole request when the body is a single object or has a `header=present` header naming it. The role is compiled its own copy of the table schema, without the columns it may not write, so the message does not confirm whether the column exists. There is no separate `403` for a denied column on insert, and the published row never carries one. +On ingest (`POST /v1/ingest?table={table}`) the same lists cap what a role may write, and a column it may not write is indistinguishable from one the table does not have: the record is refused by ClickHouse's own parser with code `117` (`Unknown field found while parsing …`), a `400` — per record in a batch, or for the whole request when the body is a single object or has a `header=present` header naming it. The role is compiled its own copy of the table schema, in which a column it may not write stays as a `MATERIALIZED` column of its default (its `DEFAULT` expression, or its type's default), so naming it is `117`, the message does not confirm whether the column exists, expressions that read it keep working, and the stored row holds the table's default for it. A role whose schema cannot be compiled this way, or that may write no column at all, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`; fix the policy or the table, a retry cannot help. There is no separate `403` for a denied column on insert, and the published row never carries one. ## Row-level security @@ -286,7 +286,7 @@ Row filters apply on the structured-query path and, per subscriber, on the live Checks are compiled into **one chtypes filter** — `col = {p:String}` for `_eq`, `col IN (…)` for `_in`, AND-joined over every checked column, with an integer column's claim compared through the strict cast described under [JWT claim templating](#jwt-claim-templating) — and evaluated against the row ClickHouse's own parser produced, by the same engine that evaluates a row `filter`. So a check sees the **stored** value, after coercion and after `DEFAULT`s: what it admits is what the table will hold. - **If the request body includes the column**, its value must satisfy the check or the record is rejected with `403 check failed for column "x"` (`… for columns "x", "y"` when more than one is checked — the filter is AND-joined, so it names the set tested rather than inventing an attribution). Comparison is ClickHouse's, under the column's own type: on an integer, `Decimal`, `UUID` or `String` column a writer cannot forge a row for another tenant, because equal values store equal. A `Float32`/`Float64` column rounds the stored value and gives that exactness back — the row can land on a neighboring id. A check the engine cannot evaluate at all is a `422`, never a silent pass. -- **If the request body omits the column** — or sends an explicit `null`, which `input_format_null_as_default` resolves the same way — an `_eq` check **auto-injects** the claim-derived value, so clients can send just the business fields and let the policy stamp `user_id` and `tenant_id` from the token. It is implemented as a `DEFAULT` on the role's compiled schema, so a value the caller *does* send still wins. An `_in` check has no single value to stamp, so the **table's own default** is what gets tested: the record is admitted if that default is in the claim-derived set and rejected if it is not. +- **If the request body omits the column** — or sends an explicit `null`, which `input_format_null_as_default` resolves the same way — an `_eq` check **auto-injects** the claim-derived value, so clients can send just the business fields and let the policy stamp `user_id` and `tenant_id` from the token. It is implemented as a `DEFAULT` on the role's compiled schema, so a value the caller *does* send still wins. An `_eq` check also covers a column the role may not otherwise write (left out of `allow_columns`, or in `deny_columns`): a record that omits it is filled with the required value, one that supplies exactly that value is accepted, and any other value fails the check (`403`). An `_in` check has no single value to stamp, so the **table's own default** is what gets tested: the record is admitted if that default is in the claim-derived set and rejected if it is not. - **The check's column must be one a record can actually carry.** A `check` naming a column the table does not have, one ClickHouse computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is refused with a per-record `403` naming the column — on *every* insert by that role, until the policy or the table is corrected. None of the three can be enforced: the published row has one slot per wire column, so an injected value for a computed or unknown column is dropped on the way out, and an ephemeral column is never stored. Each would have answered `200` while enforcing nothing. `wavehouse validate` cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**. - **A claim literal the column cannot read fails closed, not loosely.** A `_eq` value that is not a legal literal for the column (`"1.0"` on a `UInt64`) cannot be compiled as that column's `DEFAULT`, so the injection is dropped and logged; the filter then judges the record as sent, which an absent column loses. On an integer column every record then fails the check — a `403`; on another column a supplied value is ClickHouse's code 53 `TYPE_MISMATCH` per row — a `422`. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 7293fc52..48841cd7 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -254,7 +254,7 @@ The policy engine authorizes mutations by inspecting the columns being written. **What ClickHouse decides, and what WaveHouse decides.** Everything about a *value* is ClickHouse's: -- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED` and `ALIAS` columns — neither is ever part of a published row. An `EPHEMERAL` column is accepted as input wherever the format names its columns (the JSON family and the `…WithNames` formats) and feeds the `DEFAULT` expressions that read it, but it is never stored, selected or published; a positional CSV or TSV body carries the wire columns only. +- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED` and `ALIAS` columns — neither is ever part of a published row. An `EPHEMERAL` column is accepted as input only where the format names its columns (the JSON family and the `…WithNames` formats), the role may write it, a `DEFAULT` column reads it, and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column reads it; its value feeds that `DEFAULT` and is never stored, selected or published. Any other `EPHEMERAL` column is refused like an unknown one (code **117**): ClickHouse computes `MATERIALIZED` columns at insert time from the published row, which never carries the ephemeral value, so accepting it there would drop it silently. A positional CSV or TSV body (no header, or `header=absent`) carries the wire columns only. - An omitted column, or an explicit `null` on one (WaveHouse pins `input_format_null_as_default`), takes its `DEFAULT` expression — evaluated by ClickHouse, including a volatile one like `now()` — or the type's implicit zero where none is declared, exactly as an `INSERT` naming fewer columns does. - A coercion ClickHouse would make it makes here (a numeric string into an `Int*`, `"true"` into a `Bool`, an out-of-range integer wrapping); anything it would refuse fails synchronously in the ingest response with its real code, rather than surfacing later in the DLQ. `Nullable()` and `LowCardinality()` wrappers are transparent. @@ -269,6 +269,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26 or 27) | | 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | +| 400 | `{"error":"invalid json: content after the closing ']' of the json array"}` | A body declared `application/json` opening with `[` has something other than whitespace after its closing `]`. That tail is not a record of the array, so the whole request fails and nothing is published | | 400 | `{"error":"missing dedupe id field \"event_id\""}` | **Per-record.** Only with `dedupe.require_id: true`, when the row carries no value for the configured `id_field` (an absent column, a `null` cell or an empty string — the value an omitted `String` id column stores). With `require_id: false` (the default) the row is published un-deduped instead. Either way it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (the gate surfaces the token reason rather than silently falling back to `default_role`) | | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table — checked once, before any record | @@ -280,6 +281,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | | 422 | `{"error":"validation engine declined: "}` | **Per-record.** chtypes could not evaluate the record at all — the artifact declined the shape, rather than the data being wrong. A `check` clause that could not be evaluated lands here too (`validation engine declined: the insert check for column "x" could not be evaluated`) | | 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither the record's fault nor an unavailable tenant; logged. Nothing was published | +| 500 | `{"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` | The role's insert permissions do not compile against the table (for example, the role may write no column of it). It persists until the policy or the table changes, so it carries no `Retry-After` and must not be retried; the cause is in the server log | | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now: it is not open (for example, it failed to open on a reload), or a DynamoDB table is throttling, timing out or unreachable; `Retry-After: 5`. Nothing was published, so the retry is safe | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet (its ClickHouse unreachable, or [no pool for it](/settings-directory#clickhouse)), so whether the table exists is not known; `Retry-After: 5`. Decided before the body is read | @@ -303,7 +305,7 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling ClickHouse's own parser accepts under `date_time_input_format=best_effort` (the setting WaveHouse pins, both at chtypes' ingest compile and on the worker's `INSERT`) is accepted — RFC 3339 with any offset, a zone-less `YYYY-MM-DD[ T]HH:MM:SS[.fff]` read in the column's declared zone else the server's default, a Unix-seconds string, a bare integer (ClickHouse 26.8 reads a bare number in a `DateTime64` column as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — send milliseconds as a quoted string, or as a decimal number of seconds), among the other forms its lenient parser reads. It is ClickHouse's grammar, not a reimplementation of it, so whatever a real `INSERT` into this table would accept, ingest accepts, with the same coercions and the same refusals. -**Outbound**, `DateTime`/`DateTime64` values in the NATS/SSE wire row and in `/v1/query` / `/v1/pipes/{name}` results are the exact bytes ClickHouse's writer produces, as RFC 3339 in UTC, whatever zone the column declares (ClickHouse's `date_time_output_format=iso`): `"2026-06-21T04:00:00.123Z"` for a `DateTime64(3)`, `"2026-06-21T04:00:00Z"` for a `DateTime`, with the fraction at the column's own precision, trailing zeros kept. Every consumer renders from the same stored value the same way, so SSE, `/v1/query` and `/v1/pipes/{name}` agree on spelling for a given row **by construction**, with no WaveHouse rewriting step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)), and the strings order the same way the instants do. The raw-SQL proxy `/v1/ops/query` sets the same `date_time_output_format=iso`, so it spells a `DateTime`/`DateTime64` the same way; its other types are returned as ClickHouse renders them under `FORMAT JSON`. A `Date` column is `"2026-06-21"` throughout. +**Outbound**, `DateTime`/`DateTime64` values in the NATS/SSE wire row and in `/v1/query` / `/v1/pipes/{name}` results are the exact bytes ClickHouse's writer produces, as RFC 3339 in UTC, whatever zone the column declares (ClickHouse's `date_time_output_format=iso`): `"2026-06-21T04:00:00.123Z"` for a `DateTime64(3)`, `"2026-06-21T04:00:00Z"` for a `DateTime`, with the fraction at the column's own precision, trailing zeros kept (a `DateTime64(3)` on a whole second is `"…:00.000Z"`; releases before this one trimmed it to `"…:00Z"`). An event published before an upgrade to this spelling, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in. Every consumer renders from the same stored value the same way, so SSE, `/v1/query` and `/v1/pipes/{name}` agree on spelling for a given row **by construction**, with no WaveHouse rewriting step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)), and the strings order the same way the instants do. The raw-SQL proxy `/v1/ops/query` sets the same `date_time_output_format=iso`, so it spells a `DateTime`/`DateTime64` the same way; its other types are returned as ClickHouse renders them under `FORMAT JSON`. `Date`, `Date32` and `Time` columns keep their own form (`"2026-06-21"`) throughout. **One server time zone per ClickHouse line, per process.** The in-process parser takes its zone once, when a process first opens a ClickHouse line, and keeps it for the process's lifetime. A tenant whose server reports a different zone than the one that line was opened with is refused on its own — ingest answers `503`, a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working. Run tenants whose servers use different zones in separate processes. @@ -319,7 +321,7 @@ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling Cl | `text/csv; header=absent` | `CSV`, strictly positional: header detection is off (`input_format_csv_detect_header=0`), so every line is a record | | `text/csv` (no `header` parameter) | ClickHouse's default `CSV`: header auto-detection stays on | -`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. With no parameter, the positional fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column. +`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. With no parameter, the positional fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column, and only one that meets the [conditions above](#post-v1ingesttabletable--ingest-data). | Body | Outcome | | --- | --- | @@ -402,12 +404,14 @@ A `200` is returned whenever the body was read and the records were processed | 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | | 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26 or 27) | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | +| 400 | `{"error":"invalid json: content after the closing ']' of the json array"}` | A body declared `application/json` opening with `[` has something other than whitespace after its closing `]`. That tail is not a record of the array, so the whole request fails and nothing is published | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (same auth gate as the single-object path; surfaces the token reason) | | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table (checked once, before any record) | | 403 | `{"error":"insert permissions were not resolved for this request"}` | The grant that resolved for the request is not an insert grant; see the [single-record table](#post-v1ingesttabletable--ingest-data). Fails the whole request, an empty array included | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | | 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither a record's fault nor an unavailable tenant; logged. Nothing was published | +| 500 | `{"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` | The role's insert permissions do not compile against the table (for example, the role may write no column of it). It persists until the policy or the table changes, so it carries no `Retry-After` and must not be retried; the cause is in the server log | | 500 | `{"error":"publish failed"}` / `{"error":"dedupe failed"}` | Message-queue or dedup-backend failure mid-batch, other than a full queue or an unreachable broker (below). After a publish failure the records before it keep their ids, so a whole-batch retry reports those as duplicates; the failing record's id is left to lapse as on the single-object path, and the rest of its window's ids are given back | | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure) or not open, mid-batch; includes `Retry-After: 30`. The records before the refused one keep their ids, and its id and the rest of its window's are given back | | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`), mid-batch. Only under [`mq.backend: nats`](/deployment#external-nats); the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the failing record's id is left to lapse rather than given back — so `Retry-After` is that record's dedupe lease, rounded up to whole seconds, when it was deduped; a record published un-deduped has no lapsing claim to wait out, so `Retry-After: 5` | @@ -735,7 +739,7 @@ Returns the schema for a specific table. } ``` -Per-column fields: `name`, `type` and `is_nullable` describe the column; `position` is its 1-based ordinal in the table's declaration order (always present, and the order `columns` itself is in); `has_default` says whether it declares any default at all, while `default_kind` (`DEFAULT`, `MATERIALIZED`, `ALIAS`, `EPHEMERAL`) and `default_expression` say which and what — both omitted when the column declares none. A `MATERIALIZED` or `ALIAS` column is computed and **not** insertable, so it never appears in an ingest envelope or an SSE `schema` frame, though it is still reported here and can still be selected by name. An `EPHEMERAL` column is the reverse: ClickHouse accepts it in an `INSERT` — it exists to be written — and WaveHouse ingest accepts it wherever the format names columns (the JSON family and `…WithNames`), feeding the `DEFAULT` expressions that read it, but it is never stored, never selectable and never published, so it appears in no envelope, no stream column list and no query result. The table's `CREATE TABLE` statement is captured on the same refresh but is deliberately **not** exposed here — for a table backed by an external engine it renders that engine's wiring — endpoint, bucket/host, database, username, access key id. (ClickHouse masks the password as `[HIDDEN]` from ~23.9; the topology is what is withheld here.) +Per-column fields: `name`, `type` and `is_nullable` describe the column; `position` is its 1-based ordinal in the table's declaration order (always present, and the order `columns` itself is in); `has_default` says whether it declares any default at all, while `default_kind` (`DEFAULT`, `MATERIALIZED`, `ALIAS`, `EPHEMERAL`) and `default_expression` say which and what — both omitted when the column declares none. A `MATERIALIZED` or `ALIAS` column is computed and **not** insertable, so it never appears in an ingest envelope or an SSE `schema` frame, though it is still reported here and can still be selected by name. An `EPHEMERAL` column is the reverse: ClickHouse accepts it in an `INSERT` — it exists to be written — and WaveHouse ingest accepts it where the format names columns (the JSON family and `…WithNames`), the role may write it, and a `DEFAULT` expression reads it (and no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` one does), feeding that `DEFAULT`; it is never stored, never selectable and never published, so it appears in no envelope, no stream column list and no query result. The table's `CREATE TABLE` statement is captured on the same refresh but is deliberately **not** exposed here — for a table backed by an external engine it renders that engine's wiring — endpoint, bucket/host, database, username, access key id. (ClickHouse masks the password as `[HIDDEN]` from ~23.9; the topology is what is withheld here.) **Error responses:** diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index a7c7a67d..791cdb74 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -84,9 +84,9 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`sql_classify.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. -- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch, and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read and ahead of every other tenant's traffic. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). +- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (anything but whitespace after the closing `]` is a whole-request `400 invalid json: content after the closing ']' of the json array`, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read and ahead of every other tenant's traffic. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso`, the same spelling the structured-query path and the SSE wire use, so a timestamp reads the same on every surface. -- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=iso`, so a timestamp is RFC 3339 in UTC and matches the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. Each tenant's connections are capped at its `max_open_conns`, per pool identity (URL, user, database and TLS settings): two tenants that share all four share one reader and one cap, and tenants that differ in any of them never share. +- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=iso`, so a timestamp is RFC 3339 in UTC and matches the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. The read paths hold at most the tenant's pool size in HTTP connections — its `max_open_conns`, or the largest among the tenants sharing its connection tuple (address, database, user, password, TLS) — one cap per tuple, like the native pool; tenants on one server with a different database, user, password or TLS settings each have their own. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) chconn.Target` beside it — each resolves the request's tenant per call, and the zero target (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for one tenant's per-table parked counts (optionally one table) and its total — the tenant `?tenant=` names, read strictly by `opsTenant`, tenant `0` without it. The tenant is looked up in the MQ, not the settings registry, so a rejected or removed tenant's parked rows are read like a served one's; a tenant with no dead-letter queue (`mq.ErrNoDeadLetterQueue`) is a 404, and any other failure to read it a 500. The queue itself is `internal/mq`'s. @@ -158,7 +158,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `discovery/` — Schema Discovery -- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant to another user or `tls` block, which reads the same tables, applies to the next refresh, one that moves it to another address or database has `internal/app`'s `discoveries` drop its registry and build a fresh one, whose first discovery runs at once and whose lookups answer `ErrNotLoaded` until it succeeds ([#638](https://github.com/Wave-RF/WaveHouse/issues/638)), and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) describe the table's writable columns; the ingest envelope and the SSE announcement are built from the type layer's `Table.WireColumns` instead, which also leaves out `EPHEMERAL` columns. Each refresh also records the server version (`SELECT version()`) and default time zone (`SELECT timezone()`, exposed via `ServerTimezone()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), and reads each column's `default_expression` and 1-based `position` alongside its type. Overlapping refreshes (the loop and an on-demand one) publish in the order they started: a refresh that finishes after a later-started one has published returns without publishing, so an older snapshot never replaces a newer one. An `OnRefresh` hook fires synchronously after every refresh that publishes and **before** the registry reports itself loaded, so a loaded tenant is a bound one — `internal/typelayer.Engine.Bind` is its only registered consumer, and it is what resolves the chtypes artifact matching that tenant's server line and recompiles its per-table handles ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop, each sleep drawn uniformly below the backoff so instances retrying one ClickHouse do not retry in lockstep, used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. +- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant to another user or `tls` block, which reads the same tables, applies to the next refresh, one that moves it to another address or database has `internal/app`'s `discoveries` drop its registry and build a fresh one, whose first discovery runs at once and whose lookups answer `ErrNotLoaded` until it succeeds ([#638](https://github.com/Wave-RF/WaveHouse/issues/638)), and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` says which columns an `INSERT` may name (the ingest check guard reads it); which columns a record may supply, and the columns the envelope and the SSE announcement carry, are the type layer's (`typelayer.Table.WireColumns`, `typelayer.WireColumnsOf`), which also leave out `EPHEMERAL` columns. Each refresh also records the server version (`SELECT version()`) and default time zone (`SELECT timezone()`, exposed via `ServerTimezone()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), and reads each column's `default_expression` and 1-based `position` alongside its type. Overlapping refreshes (the loop and an on-demand one) publish in the order they started: a refresh that finishes after a later-started one has published returns without publishing, so an older snapshot never replaces a newer one. An `OnRefresh` hook fires synchronously after every refresh that publishes and **before** the registry reports itself loaded, so a loaded tenant is a bound one — `internal/typelayer.Engine.Bind` is its only registered consumer, and it is what resolves the chtypes artifact matching that tenant's server line and recompiles its per-table handles ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop, each sleep drawn uniformly below the backoff so instances retrying one ClickHouse do not retry in lockstep, used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. - **discovery_test.go** — Unit tests for schema discovery. ### `typelayer/` — In-Process ClickHouse Parser @@ -168,11 +168,12 @@ The only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One proce Each tenant has its own table set inside that engine, bound from the tenant's own schema refresh and released when the tenant is no longer served. - **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind. A tenant whose ClickHouse line has no installed artifact is unavailable on its own, and so is one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Either way the cause is recorded (an ingest client sees only a generic `503`; the cause goes to the log), every other tenant keeps working, and a later bind of the same tenant (a reload, a refresh that now agrees) clears it. -- **`Engine.RoleTable(tenant, table, shape)`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column is simply not in the schema, so naming it is ClickHouse's code 117, and an absent check column takes the claim as its default while a supplied value still wins. +- **`Engine.RoleTable(tenant, table, shape)`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column, including one the role may not otherwise write — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column stays in the schema as `MATERIALIZED` of its default, so naming it is ClickHouse's code 117 while expressions that read it still compile, and an absent check column takes the claim as its default while a supplied value is accepted only if it equals the claim. A role-shape pool starts at one handle and grows to `min(GOMAXPROCS, 4)` when every handle is busy; a base table's grows to `min(GOMAXPROCS, 8)`. - **`Table.IngestWith(format, opts, body, checks...)`** (`Ingest` is the same with no options) runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. - **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per table. Only a definite true admits; a predicate error, a policy column the table no longer has, schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. - A tenant that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row of that tenant's tables from a role that has a row `filter`, with reason `unavailable`. -- A role whose column policy cannot be compiled is refused with a non-retryable error and the cause is logged; it is not the `503` an unavailable tenant gets, since retrying cannot fix a policy. +- A role whose schema cannot be compiled — or that may write no column of the table — is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`, and the cause is logged once a minute; it is not the `503` an unavailable tenant gets, since retrying cannot fix a policy or a table. +- **`typelayertest`** (`internal/typelayer/typelayertest`) holds the test helpers other packages' tests use — `TestEngine`, `SkipWithoutArtifact`, `RequireEnv` — which only test binaries link (`typelayer` itself never imports `testing`). The `typelayer` package's own tests use a small in-package copy, since they cannot import a package that imports them. See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest error-response shape and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced) for how predicates are compiled and evaluated. @@ -284,8 +285,8 @@ Client POST /v1/ingest?table={table} (including now()) per record. A rejected record carries ClickHouse's real error code (exception_code) and message — an unknown field, a MATERIALIZED/ALIAS column, or a column this role may not write is - 117 (an EPHEMERAL value is accepted where the format names columns and - feeds DEFAULTs, never stored or published); a record chtypes cannot answer for is a distinct "declined" outcome + 117 (an EPHEMERAL value is accepted only where the format names columns, the + role may write it and a DEFAULT reads it; it feeds that DEFAULT, never stored or published); a record chtypes cannot answer for is a distinct "declined" outcome (422), never a data rejection → Evaluate the role's check clauses over the accepted rows with one compiled chtypes filter — false is 403 for that record, unevaluable is 422 diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index 53429b57..b189d545 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -167,7 +167,7 @@ Only the secret, the connection ceiling and the chtypes artifact directory are b | YAML Key | Env Var | Default | Description | | --- | --- | ------- | ----------- | | `clickhouse.password` | `WH_CH_PASSWORD` | *(empty)* | Authentication password, combined with the settings directory's `clickhouse.username` on every (re)connect. A secret, so it never lives in a tracked JSON file; rotating it is a restart. | -| `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the `max_open_conns` of the open pools — one per distinct connection tuple among the served tenants, see [the settings directory](/settings-directory#clickhouse) — must not add up to more. Pools above it at boot refuse to start, naming the sum and the ceiling; on a reload a pool resized above it is refused and keeps its size, and a new pool that would cross it is not opened — its tenants keep the pool they had, or have none — both logged (the reload itself still reports `adopted`) and retried by the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. It counts native connections only: the HTTP reader behind structured queries and pipes is capped separately, per pool identity (URL, user, database and TLS settings), by that pool's `max_open_conns`. | +| `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the `max_open_conns` of the open pools — one per distinct connection tuple among the served tenants, see [the settings directory](/settings-directory#clickhouse) — must not add up to more. Pools above it at boot refuse to start, naming the sum and the ceiling; on a reload a pool resized above it is refused and keeps its size, and a new pool that would cross it is not opened — its tenants keep the pool they had, or have none — both logged (the reload itself still reports `adopted`) and retried by the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. It counts native connections only: the HTTP reader behind structured queries and pipes is capped separately, one cap per connection tuple (URL, user, password, database, TLS), at the largest `max_open_conns` among the tenants sharing it. | | `clickhouse.chtypes_registry` | `WH_CHTYPES_REGISTRY` | *(empty)* | Directory holding the chtypes artifacts (one `/` per ClickHouse line). Empty defers to the chtypes search path — `$CHTYPES_REGISTRY`, `~/.cache/chtypes/artifacts/abi6/-`, then the system directories. Either way a library is opened lazily, on first use of its line; an explicit directory is searched first, then the rest of the path. Boot-tier: changing it is a restart. See [chtypes artifacts](/deployment#chtypes-artifacts). | ### ClickHouse user access diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index baa2ccf5..79b44d26 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -645,7 +645,7 @@ For local development, `docker compose -f deployments/compose/dependencies.yaml WaveHouse uses a **Bring Your Own Schema** model. You create your tables in ClickHouse with whatever columns and engines you need. WaveHouse discovers the schemas automatically via `system.columns` and validates ingest data against them — see [Schema Validation](/api#post-v1ingesttabletable--ingest-data) for the rules a record must satisfy. -Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted where the format names columns (the JSON family, `…WithNames`) and feeds the `DEFAULT`s that read it without being stored or published; a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's implicit zero value where none is declared, evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). +Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`/`ALIAS` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's implicit zero value where none is declared, evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). Example table: diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index b26ec107..83edc082 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -357,7 +357,7 @@ Each test target writes `covdata` to `tmp/coverage//data/`, renders a tex - **Unit tests** live beside the code they test (e.g., `internal/discovery/discovery_test.go`). They use mocks or embedded NATS (in-process, no Docker needed). - **Integration tests** use the `//go:build integration` build tag. In `tests/integration`, `TestMain` starts one ClickHouse testcontainer and a dynamodb-local one (for the DynamoDB dedupe backend's tests), and boots the production wiring against it through `app.New` (embedded NATS, ingest worker, sweeper, hub, the API server on a random loopback port); tests reach it via `env(t)` and create their own tables. `TestNATSBackend_EndToEnd` also starts a NATS container configured from `deployments/nats/values.yaml`, applies `deployments/nats/jetstream.yaml` to it through `internal/mq/natstest`, and boots two processes on `mq.backend: nats` against it. `TestMain` also builds the `wavehouse` binary while the containers start, so the build is not charged to the `-timeout`: the `TestRoles_*` tests run it as separate OS processes (two API, two then three ingest, one killed with `SIGKILL`) over NATS, Redis and dynamodb-local, and read each ingest process's shard ownership from its `/metrics`. DLQ tests use `assert.Eventually` with a 30-second timeout for the 5-second ingest worker batch window. `internal/cache`'s integration tests start their own containers instead — Redis, Valkey, Dragonfly and a one-node Redis Cluster — for the shared backend. `shared_cache_test.go` starts its own Redis testcontainer per test (`startRedis`) and boots extra, independent `cache.backend: redis` instances over that same ClickHouse (`bootRedisApp`), to exercise the cache shared across processes rather than one package in isolation. The same target also runs `internal/mq/natsspike`. That package pins the nats-server behavior the external-NATS topology depends on, against an in-process server with no Docker. It lives under `internal/mq` because only that tree may import NATS, and it runs here rather than in the unit suite because each test takes seconds and the unit suite has a 15-second limit per package. For the same reason the external NATS broker's tests (`internal/mq/external*_test.go`, including its run of the `mqtest` conformance suite) and the NATS KV lease tests (`internal/mq/lease_test.go`, including their run of the `coordtest` conformance suite) carry the `integration` tag inside `internal/mq`, and the target runs them by name, so the package's untagged tests stay in the unit suite alone. -Shared test utilities live in `internal/testutil/`. The packages log through `slog.Default()`, so tests reach log output through `internal/testutil/logtest`: `logtest.Silence()` in a package's `TestMain` discards it, and `logtest.Capture(t, level)` routes it to a buffer for a test that asserts on log lines — such a test must not call `t.Parallel()`, because the default logger is process-wide. A test that starts the embedded broker keeps its store in `internal/testutil/storedir`'s `storedir.New(t)` rather than a bare `t.TempDir()` (`testutil.NewEmbeddedMQ` does): the NATS server can finish writing a consumer's state after `Close` returns, which fails `t.TempDir`'s one-shot removal, and `storedir` removes the store again until those writes have landed ([#442](https://github.com/Wave-RF/WaveHouse/issues/442)). +Shared test utilities live in `internal/testutil/`. The packages log through `slog.Default()`, so tests reach log output through `internal/testutil/logtest`: `logtest.Silence()` in a package's `TestMain` discards it, and `logtest.Capture(t, level)` routes it to a buffer for a test that asserts on log lines — such a test must not call `t.Parallel()`, because the default logger is process-wide. A test that needs the chtypes artifact takes its engine from `internal/typelayer/typelayertest` (`typelayertest.TestEngine`, and `typelayertest.SkipWithoutArtifact` to skip when the artifact is not installed — set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` to fail instead); `internal/typelayer`'s own tests use an in-package copy of that helper. A test that starts the embedded broker keeps its store in `internal/testutil/storedir`'s `storedir.New(t)` rather than a bare `t.TempDir()` (`testutil.NewEmbeddedMQ` does): the NATS server can finish writing a consumer's state after `Close` returns, which fails `t.TempDir`'s one-shot removal, and `storedir` removes the store again until those writes have landed ([#442](https://github.com/Wave-RF/WaveHouse/issues/442)). ### Adding New Tests diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index 96b6bcdb..b2811958 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -147,7 +147,7 @@ Codegen reads `/v1/ops/schema`, which is **admin-only**. Against a non-dev serve | `--out`, `-o` | Output .d.ts file path | `./wavehouse.d.ts` | | `--auth`, `-a` | Bearer token (if auth required) | — | -The generated row type is the **read** shape, and computed columns are where it and the server disagree. An `EPHEMERAL` column declares a default, so codegen emits it, yet no query can ever return it — the type says readable where only the write is real. `MATERIALIZED` and `ALIAS` columns declare defaults too, so they are emitted as optional, but supplying either on `insert` is a `400` carrying ClickHouse's own code 117 (`Unknown field found while parsing JSONEachRow format: x`), and the type will not catch it. An `EPHEMERAL` value is accepted on a JSON `insert` and feeds the defaults that read it, but is never stored or returned. Omit computed columns; the server fills them in. +The generated row type is the **read** shape, and computed columns are where it and the server disagree. An `EPHEMERAL` column declares a default, so codegen emits it, yet no query can ever return it — the type says readable where only the write is real. `MATERIALIZED` and `ALIAS` columns declare defaults too, so they are emitted as optional, but supplying either on `insert` is a `400` carrying ClickHouse's own code 117 (`Unknown field found while parsing JSONEachRow format: x`), and the type will not catch it. An `EPHEMERAL` value is accepted on a JSON `insert` when the role may write the column, a `DEFAULT` column reads it and no `MATERIALIZED` or `ALIAS` column does; it feeds that default and is never stored or returned. Otherwise it is a `400` with code 117, like an unknown column. Omit computed columns; the server fills them in. **Example output:** diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index 2c803df5..2d56ba43 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -99,7 +99,7 @@ interface StreamEvent { `data` is a row **object**, as it always has been — but the wire underneath is positional. The server sends the column list in its own `event: schema` frame — before the first row, and again whenever the list drifts on the **live** path (with one exception after a gap-fill, below) — and each row as a JSON array; the SDK keeps the announced list and zips every row against it, so this shape is unchanged and nothing in your code moves. It matters in two places. The row object has a **null prototype**: a ClickHouse column may legitimately be named `__proto__`, and on an ordinary object that assignment hits the inherited setter and the value disappears — so the SDK builds each row with `Object.create(null)`. Property access, spreading, `JSON.stringify` and destructuring all behave normally; what does not is anything inherited from `Object.prototype`, so use `Object.hasOwn(row, "x")` rather than `row.hasOwnProperty("x")`, and don't rely on `` `${row}` `` or `row.constructor`. (`liveQuery`'s REST backfill half still yields ordinary objects.) And a **raw** SSE consumer (a hand-rolled `EventSource`) must do the zipping itself — see [the wire format](/api#get-v1stream--server-sent-events-stream). A column the producer omitted is no longer `null` on the wire: WaveHouse's ingest validation runs ClickHouse's own parser in-process, which evaluates the column's `DEFAULT` (or its implicit zero value) before the row is published — the same as a native `INSERT` naming fewer columns than the table has. -Row values of top-level `DateTime`/`DateTime64` columns inside `data` (not timestamps nested in `Array`/`Map`/`Tuple` columns) arrive as RFC 3339 in UTC, whatever zone the column declares — `"2026-06-21T04:00:00.123Z"` — matching what `/v1/query` returns for the same row byte-for-byte, by construction (see [Timestamp rendering](/api#timestamp-rendering)). `new Date(value)` parses it directly, and because the stream's `timestamp` field and a timestamp column use the same form, the SDK's comparisons between them (the live-query dedupe below, client-side `.where()` on a timestamp column) are comparisons between like spellings. +Row values of top-level `DateTime`/`DateTime64` columns inside `data` (not timestamps nested in `Array`/`Map`/`Tuple` columns) arrive as RFC 3339 in UTC, whatever zone the column declares — `"2026-06-21T04:00:00.123Z"` — matching what `/v1/query` returns for the same row byte-for-byte, by construction (see [Timestamp rendering](/api#timestamp-rendering)). `new Date(value)` parses it directly, and because the stream's `timestamp` field and a timestamp column use the same form, the SDK's comparisons between them (the live-query dedupe below, client-side `.where()` on a timestamp column) are by instant, not by spelling. A `DateTime64(3)` on a whole second arrives as `…:00.000Z`, and an event published before an upgrade, or by an older instance during a rolling deploy, replays in the spelling it was published in. ### Transport Behavior @@ -150,6 +150,8 @@ const stream = wh.from('clicks') Supported operators: `=`, `!=`, `>`, `>=`, `<`, `<=`, `in`, `like`, `not_like` — the same `FilterOp` set `.where()` takes everywhere (the SDK maps them to wire tokens such as `eq`/`neq` internally). +Client-side `where` filters compare two timestamps as instants, to the nanosecond (`=`, `!=`, `in`, `>`, `>=`, `<`, `<=`), so `.where('received_timestamp', '>', '2026-03-24T12:00:00Z')` matches `2026-03-24T12:00:00.001Z` and `=` matches `…00.000Z`. A value with no zone is read as UTC (the server reads it in the column's zone). Other strings compare as strings. + Two of them need care, for different reasons. `like` matches **case-insensitively** here, while ClickHouse's `LIKE` on the query path is case-sensitive — so inside a single `liveQuery()` the backfill rows arrive through the server's case-sensitive filter and the live frames through this one, and a pattern like `'/Home%'` can admit live events whose historical counterparts the backfill excluded. `not_like` has no server-side counterpart at all: `/v1/query` rejects it with a `400` (see the operator table in [Queries](/sdk/queries)), so a `liveQuery()` filtered on it gets an error `Result` in `initial()`, drops whatever was buffered during the backfill window, and runs live-only from there. --- @@ -195,7 +197,7 @@ interface StreamSubscriber { 1. Opens the stream **immediately** and buffers incoming events. 2. Runs the `.fetch()` query for historical data, calls `subscriber.initial()` with the result — unless the fetch itself throws, in which case neither happens (see below). -3. Deduplicates buffered events against the **last row** of the backfill result — which is the newest only when the query orders ascending. A `desc` query (like the example above) puts the oldest row last, so for live frames the boundary is the oldest timestamp and this step filters nothing. With a `since` gap-fill in flight it works the other way — replay lands in the same buffer, so replayed events at or older than that row are dropped before reaching `next()`. A projection that omits `received_timestamp` skips the pass entirely. Tracked in [#449](https://github.com/Wave-RF/WaveHouse/issues/449). +3. Deduplicates buffered events against the **last row** of the backfill result — which is the newest only when the query orders ascending. A `desc` query (like the example above) puts the oldest row last, so for live frames the boundary is the oldest timestamp and this step filters nothing. With a `since` gap-fill in flight it works the other way — replay lands in the same buffer, so replayed events at or older than that row are dropped before reaching `next()`. The comparison is by instant, at the coarser of the two precisions, so an event in the boundary row's own millisecond counts as covered. A projection that omits `received_timestamp` skips the pass entirely. Tracked in [#449](https://github.com/Wave-RF/WaveHouse/issues/449). 4. Flushes remaining buffered events and switches to live mode. This "stream-first" approach is what closes the window between the fetch and the stream starting — events arriving during the fetch are buffered rather than missed. It holds only when the backfill completes cleanly. diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 9d369547..6f418704 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -116,7 +116,7 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `clickhouse.tls.insecure_skip_verify` | `false` | Skips server certificate verification. `true` validates with a warning: both hops then accept any certificate. | | `clickhouse.tls.server_name` | `""` | Name the server certificate is verified against; empty derives it from the host in `addr` (native) or the URL (HTTP). | | `clickhouse.headers` | `{}` | Extra request headers for the HTTP interface (structured queries, pipes, ingest `INSERT`s, `POST /v1/ops/query`). WaveHouse's own `Content-Type` and credential headers are set after them and win; naming `X-ClickHouse-User`, `X-ClickHouse-Key` or `Authorization` is a validation error, and so are two spellings of one name (names are case-insensitive). See [ClickHouse](#clickhouse). | -| `clickhouse.max_open_conns` | `10` | Native connection pool size (`>= max_idle_conns`); tenants sharing a pool size it to the largest ask among them. It also caps the HTTP reader connections behind structured queries and pipes, per pool identity (URL, user, database and TLS settings). The open pools together must not exceed the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) when one is set. See [ClickHouse](#clickhouse). | +| `clickhouse.max_open_conns` | `10` | Native connection pool size (`>= max_idle_conns`); tenants sharing a pool size it to the largest ask among them. It also caps the HTTP connections structured queries and pipes hold, one cap per connection tuple (URL, user, password, database, TLS), sized to the largest `max_open_conns` among the tenants sharing it. The open pools together must not exceed the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) when one is set. See [ClickHouse](#clickhouse). | | `clickhouse.max_idle_conns` | `5` | Idle native connections kept open (`>= 1`). | | `auth.jwks_url` | `""` | JWKS endpoint (absolute `http(s)` URL). When set, JWKS is the **sole** verifier and `jwt_secret` is ignored. See [Authentication](#authentication). | | `auth.role_claim` | `role` | Dot-separated JWT claim path the role is read from (e.g. `app_metadata.role`). | From 9dde7edf2d955fa284f81b7d872d3f8e6adf1b15 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 08:35:41 -0400 Subject: [PATCH 42/70] docs: fix round-2 review findings (RFC 3339 examples, wire columns, retryable 500, fetch step) Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- AGENTS.md | 6 +++--- README.md | 1 + docs/src/content/docs/access-control.mdx | 2 +- docs/src/content/docs/api.md | 12 ++++++------ docs/src/content/docs/architecture.md | 4 ++-- docs/src/content/docs/deployment.md | 4 ++-- docs/src/content/docs/development.md | 6 +++--- docs/src/content/docs/sdk/reference.md | 6 +++--- 8 files changed, 21 insertions(+), 20 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 74126f00..f88eb683 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -72,13 +72,13 @@ The invariant index — what must stay true. Full narrative and rationale live i 16. **Bearer-token-only CORS posture (security)** — Bearer JWT on every request, no cookies/sessions; `corsMiddleware` deliberately **never** emits `Access-Control-Allow-Credentials` (not needed, and `*` + credentials is a spec violation browsers reject). `cors.allowed_origins` (settings directory, per tenant: a tenant route is decorated from the list of the tenant it names, everything else from tenant `0`'s — `corsOrigins`) controls who can *read* responses, not cookie scope; CSRF protection is structural. Don't reintroduce cookie auth or `Allow-Credentials` without a design discussion — answers GitHub #29/#30. Code: `internal/api/router.go`. 17. **Non-fatal boot** — schema-discovery failure on boot is non-fatal: `internal/app` records an `api.BootState`, binds `:8080`, serves 503 on `/livez`/`/readyz` with the diagnostic, and retries via `SchemaRegistry.RetryRefresh` (jittered backoff 2s → 60s), per tenant over a nested directory: `/livez` is 503 while no tenant has completed a first discovery, then sticky 200, and a tenant's outage after that is its log line and counter, never a probe failure. Until a tenant's first discovery its table lookups are a 503 with `Retry-After`, not a 404. Bounds supervisor restart loops. 18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention; `/readyz` pings every open ClickHouse pool at once and is ready at the first answer, 503 naming each when none answers); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. -19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), in the column's declared zone else the server's default (`"2026-06-21 04:00:00.123"`, never RFC 3339's `Z` suffix), and `/v1/query` and pipes are rendered by the same ClickHouse in the same spelling — there is no separate WaveHouse rewrite step to keep in sync, so live and query reads can't drift on spelling *or* instant (#372) while the process's chtypes zone equals the zone the server renders in. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. +19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), as RFC 3339 in UTC at the column's scale (`date_time_output_format=iso`, e.g. `"2026-06-21T04:00:00.123Z"`), the spelling `/v1/query` and pipes pin too — there is no separate WaveHouse rewrite step to keep in sync, so live and query reads can't drift on spelling *or* instant (#372), whatever zone the column or server uses. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. 20. **Sealed MQ boundary** — only `internal/mq` imports NATS/JetStream (`github.com/nats-io/…`), enforced by the `depguard` rule in `.golangci.yml`, so `make lint` fails on a leak in every package it builds (the `integration`-tagged files under `tests/` are outside lint's build context — keep them clean by convention, through `mq.Broker`). A test outside `internal/mq` that needs a real NATS server goes through `internal/mq/natstest`, which stands one up from the shipped `deployments/nats` files and hands back a URL and passwords, never a NATS type. The boundary is semantic as well: everything else addresses events by `mq.Topic` and states intent through mq-owned interfaces (`Publisher`, `Consumer`, `DeadLetterer`, `Purger`, `Replayer`, …), and never builds a subject, names a stream, or reasons in sequences — so a subject, stream, or broker change lands in one package ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 4; story 5's tenant token landed there alone — `Topic.Tenant`, first in every subject). Don't add a raw accessor (`JetStream()`, `NatsConn()`, `GetServer()`) back, and don't hand-build `"ingest."`/`"dlq."` subjects outside `internal/mq` — widen the mq surface with an intent-level method instead. -21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503` (generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds every row with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. +21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503` (generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds every row of a role that has a row `filter` with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and, on Linux, glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. ## Code Conventions -- **Go 1.27**, strict formatting (`gofumpt`, enforced by CI); cgo enabled (`internal/typelayer`'s chtypes dlopen shim needs a C toolchain + glibc) +- **Go 1.27**, strict formatting (`gofumpt`, enforced by CI); cgo enabled (`internal/typelayer`'s chtypes dlopen shim needs a C toolchain and, on Linux, glibc) - **Structured logging** with `log/slog` (JSON handler), through the default logger: call `slog.InfoContext(ctx, …)` and its siblings (the context carries the trace ids the handler stamps) rather than taking a `*slog.Logger` parameter or field. `cmd/wavehouse` and `internal/app` install the default; tests silence or capture it with `internal/testutil/logtest` (a capturing test must not call `t.Parallel()`) - **Chi v5** for HTTP routing - **Error handling**: Return errors, don't panic. Wrap with `fmt.Errorf("context: %w", err)`. diff --git a/README.md b/README.md index 2edbc80b..e9432aac 100644 --- a/README.md +++ b/README.md @@ -158,6 +158,7 @@ You'll need **Go 1.27+, GNU Make 4+, Docker (Compose v2), Node.js 22 LTS, and pn ```bash make tools # one-time bootstrap +scripts/fetch-chtypes.sh # once per machine: the chtypes artifact (160–290 MB) docker compose -f deployments/compose/dependencies.yaml up -d clickhouse make dev # hot-reload on .go save ``` diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 4aefaf3c..879bdb6a 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -192,7 +192,7 @@ The rules, in order: On a structured query (`POST /v1/query?table={table}`) the allowlist is a **hard cap on every column the query references — in any clause**: the projection, an aggregation argument, `filters`, `group_by`, `order_by`, and `time_range`. Naming a disallowed column anywhere is rejected with `403 column "x" not allowed`. A full-row read is requested explicitly with `"select_all": true`, which expands to exactly the columns the role may read — never a raw `SELECT *` that could include a denied column; if the role is allowed *no* columns, the read is rejected (`403`) rather than returning empty rows. **Omitting `columns` (or sending `[]` / `""`) returns nothing** — a request for no data — so a hidden column can't leak by being left out, grouped on, or filtered on to infer its values. (Note: in a query, `["*"]` is the *literal column named `*`*, not a wildcard — use `select_all` for all columns. In `allow_columns`, `["*"]` is still the all-columns wildcard.) On live streams, denied columns are silently **stripped** from each event rather than rejecting the connection. The structured-query and live-stream paths defer to the **same** per-column decision (`IsColumnAllowed`), so the two read surfaces enforce identical column visibility and can't drift apart. -On ingest (`POST /v1/ingest?table={table}`) the same lists cap what a role may write, and a column it may not write is indistinguishable from one the table does not have: the record is refused by ClickHouse's own parser with code `117` (`Unknown field found while parsing …`), a `400` — per record in a batch, or for the whole request when the body is a single object or has a `header=present` header naming it. The role is compiled its own copy of the table schema, in which a column it may not write stays as a `MATERIALIZED` column of its default (its `DEFAULT` expression, or its type's default), so naming it is `117`, the message does not confirm whether the column exists, expressions that read it keep working, and the stored row holds the table's default for it. A role whose schema cannot be compiled this way, or that may write no column at all, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`; fix the policy or the table, a retry cannot help. There is no separate `403` for a denied column on insert, and the published row never carries one. +On ingest (`POST /v1/ingest?table={table}`) the same lists cap what a role may write, and a column it may not write is indistinguishable from one the table does not have: the record is refused by ClickHouse's own parser with code `117` (`Unknown field found while parsing …`), a `400` — per record in a batch, or for the whole request when the body is a single object or has a `header=present` header naming it. The role is compiled its own copy of the table schema, in which a column it may not write stays as a `MATERIALIZED` column of its default (its `DEFAULT` expression, or its type's default), so naming it is `117` (unless an `_eq` check covers it, see [Insert checks](#insert-checks)), the message does not confirm whether the column exists, expressions that read it keep working, and the stored row holds the table's default for it. A role whose schema cannot be compiled this way, or that may write no column at all, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`; fix the policy or the table, a retry cannot help. There is no separate `403` for a denied column on insert, and the published row never carries one, unless an `_eq` check covers it (see [Insert checks](#insert-checks)). ## Row-level security diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 48841cd7..d4626de2 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -254,7 +254,7 @@ The policy engine authorizes mutations by inspecting the columns being written. **What ClickHouse decides, and what WaveHouse decides.** Everything about a *value* is ClickHouse's: -- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED` and `ALIAS` columns — neither is ever part of a published row. An `EPHEMERAL` column is accepted as input only where the format names its columns (the JSON family and the `…WithNames` formats), the role may write it, a `DEFAULT` column reads it, and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column reads it; its value feeds that `DEFAULT` and is never stored, selected or published. Any other `EPHEMERAL` column is refused like an unknown one (code **117**): ClickHouse computes `MATERIALIZED` columns at insert time from the published row, which never carries the ephemeral value, so accepting it there would drop it silently. A positional CSV or TSV body (no header, or `header=absent`) carries the wire columns only. +- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED` and `ALIAS` columns — neither is ever part of a published row. (An `_eq` [insert check](/access-control#insert-checks) on a column the role may not write is the one exception: it accepts exactly the required value, and the row carries it.) An `EPHEMERAL` column is accepted as input only where the format names its columns (the JSON family and the `…WithNames` formats), the role may write it, a `DEFAULT` column reads it, and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column reads it; its value feeds that `DEFAULT` and is never stored, selected or published. Any other `EPHEMERAL` column is refused like an unknown one (code **117**): ClickHouse computes `MATERIALIZED` columns at insert time from the published row, which never carries the ephemeral value, so accepting it there would drop it silently. A positional CSV or TSV body (no header, or `header=absent`) carries the wire columns only. - An omitted column, or an explicit `null` on one (WaveHouse pins `input_format_null_as_default`), takes its `DEFAULT` expression — evaluated by ClickHouse, including a volatile one like `now()` — or the type's implicit zero where none is declared, exactly as an `INSERT` naming fewer columns does. - A coercion ClickHouse would make it makes here (a numeric string into an `Int*`, `"true"` into a `Bool`, an out-of-range integer wrapping); anything it would refuse fails synchronously in the ingest response with its real code, rather than surfacing later in the DLQ. `Nullable()` and `LowCardinality()` wrappers are transparent. @@ -321,7 +321,7 @@ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling Cl | `text/csv; header=absent` | `CSV`, strictly positional: header detection is off (`input_format_csv_detect_header=0`), so every line is a record | | `text/csv` (no `header` parameter) | ClickHouse's default `CSV`: header auto-detection stays on | -`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. With no parameter, the positional fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column, and only one that meets the [conditions above](#post-v1ingesttabletable--ingest-data). +`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. With no parameter, the positional fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column, and, for a role with column restrictions, minus every column it may not write (an `_eq`-checked column stays) — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column, and only one that meets the [conditions above](#post-v1ingesttabletable--ingest-data). | Body | Outcome | | --- | --- | @@ -421,7 +421,7 @@ A `200` is returned whenever the body was read and the records were processed | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] -A batch aborted partway — a `503` or `500` after some leading records were already published — re-publishes those leading records when the whole batch is retried. Records are published in windows of 256, in order: a dedupe failure drops the open window unpublished, so what an aborted batch published is the windows before it, plus, after a publish failure, the records of its window before the failing one. Failures decided before any record is processed are not in this class: a `413`, a `415`, the `400 invalid request body` of an upload cut off in transit, and the unterminated-array `400` all publish nothing and are safe to retry as-is (once split, for a `413`). Enable deduplication if duplicate suppression matters — the single-object path has the same at-least-once property, and the SDK retries both on `503`. +A batch aborted partway — a `503` or `500` after some leading records were already published — re-publishes those leading records when the whole batch is retried. Records are published in windows of 256, in order: a dedupe failure drops the open window unpublished, so what an aborted batch published is the windows before it, plus, after a publish failure, the records of its window before the failing one. Failures decided before any record is processed are not in this class: a `413`, a `415`, the `400 invalid request body` of an upload cut off in transit, the unterminated-array `400` and the `400 invalid json: content after the closing ']' of the json array` all publish nothing and are safe to retry as-is (once split, for a `413`). Enable deduplication if duplicate suppression matters — the single-object path has the same at-least-once property, and the SDK retries both on `503`. ::: --- @@ -663,10 +663,10 @@ event: schema data: {"table_name":"clicks","columns":["page","button","score","received_timestamp"]} id: 2026-03-24T12:00:00.123Z -data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:00.123Z","row":["/home","signup",42.5,"2026-03-24 11:59:58.512"]} +data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:00.123Z","row":["/home","signup",42.5,"2026-03-24T11:59:58.512Z"]} id: 2026-03-24T12:00:01.456Z -data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,"2026-03-24 12:00:01.456"]} +data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,"2026-03-24T12:00:01.456Z"]} ``` A raw consumer must keep the most recent announced column list and zip each `row` against it; a value the record did not carry arrives as `null` in its slot rather than being omitted, so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you and still yields row objects — `.stream()` and `.liveQuery()` are unchanged. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. @@ -888,7 +888,7 @@ The message format used on NATS JetStream between ingest and the batch consumer: "received_timestamp": "2026-03-24T12:00:00.123456789Z", "format": "JSONCompactEachRow", "columns": ["page", "button", "score", "received_timestamp"], - "row": ["/home", "signup", 42.5, "2026-03-24 12:00:00.123"] + "row": ["/home", "signup", 42.5, "2026-03-24T12:00:00.123Z"] } ``` diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 791cdb74..b5b20c22 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -179,7 +179,7 @@ See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest e ### `ingest/` — Ingest Pipeline, DLQ & Sweeping -- **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line — the exact bytes ClickHouse's own writer produced for the stored record — with `columns` naming its positions (the table's insertable columns, or the narrower list a column-restricted role produced); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with `typelayer.InsertSettings()` (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) plus `async_insert=0`, the same parsing settings chtypes compiled the row with. The worker never loads the artifact: `InsertSettings` is static. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so under `mq.backend: embedded` the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Under `mq.backend: nats`, anyone the operator lets publish to `.ingest.>` reaches it too, past auth, policy and schema validation, so that right belongs to the `wavehouse` user alone. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy — or through an operator-authored [pipe that writes](/pipes#pipes-that-write), gated only by its `allowed_roles`. A batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling) is never tried: no row of it could pass, so `parkBatch` takes it to the DLQ switch whole, logging once per batch rather than twice per row. Otherwise a bulk-insert failure is first classed by `chconn.Classify`: a ClickHouse that cannot take the insert (unavailable, denied, or no verdict at all) sends the batch back to the MQ for a delayed redelivery (`retryLater` → `mq.Message.NakWithDelay`), under a backoff shared by every table on the same pool (a failure of one table — read-only, too many parts — backs off that table alone), and never to the DLQ — the same when it stops answering mid-isolation. Only when ClickHouse rejects the batch, or refuses a multi-row batch for its size (`chconn.Splittable`: too many partitions for one INSERT, the memory limit), is it re-inserted row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. +- **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line — the exact bytes ClickHouse's own writer produced for the stored record — with `columns` naming its positions (the table's wire columns, `typelayer.Table.WireColumns`: no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` column — or the narrower list a column-restricted role produced); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with `typelayer.InsertSettings()` (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) plus `async_insert=0`, the same parsing settings chtypes compiled the row with. The worker never loads the artifact: `InsertSettings` is static. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so under `mq.backend: embedded` the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Under `mq.backend: nats`, anyone the operator lets publish to `.ingest.>` reaches it too, past auth, policy and schema validation, so that right belongs to the `wavehouse` user alone. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy — or through an operator-authored [pipe that writes](/pipes#pipes-that-write), gated only by its `allowed_roles`. A batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling) is never tried: no row of it could pass, so `parkBatch` takes it to the DLQ switch whole, logging once per batch rather than twice per row. Otherwise a bulk-insert failure is first classed by `chconn.Classify`: a ClickHouse that cannot take the insert (unavailable, denied, or no verdict at all) sends the batch back to the MQ for a delayed redelivery (`retryLater` → `mq.Message.NakWithDelay`), under a backoff shared by every table on the same pool (a failure of one table — read-only, too many parts — backs off that table alone), and never to the DLQ — the same when it stops answering mid-isolation. Only when ClickHouse rejects the batch, or refuses a multi-row batch for its size (`chconn.Splittable`: too many partitions for one INSERT, the memory limit), is it re-inserted row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. - **backoff.go** — The retry backoff behind `retryLater`: a small circuit breaker per ClickHouse pool (the target's URL, user and database), and one per pool and table for a failure of one table (`chconn.TableScoped`). A failure opens it for 1 s, doubling to a 30 s cap, each window jittered down to half; while it is open, flushes and arriving rows are handed back without a request, and once it elapses one flush probes. Any answer that is not an outage closes it. - **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. `Row` is the exact bytes `internal/typelayer`'s `IngestWith` returned for an accepted record — ClickHouse's own `JSONCompactEachRow` writer output, `DEFAULT`s already filled in — not a value WaveHouse encodes itself. - **claims.go**, **assign.go** — `ClaimShards` wraps a `Sharded` queue so an ingest process consumes only its share of the units, one process per unit at a time. Each process holds one membership lease, `ingest.m` for the lowest free `j` under the number of configured units (at most 64, read 16 at a time each tick), and every tick (2s) reads which slots are live (`coord.Observer.Held`); `assignUnits` gives every unit to a live slot by rendezvous hashing capped at ⌈units/slots⌉, the same result in every process for the same view; the extras are assigned apart from the configured units, so processes whose lists of extras differ (each refreshes it every five minutes) still agree on every configured unit. A unit no longer assigned halts (`mq.Halter`: stops fetching, keeps its pin, returns once what it fetched reached the worker), waits (bounded by the worker's 60s ack wait, the shortest `ack_wait` a durable may have) for the rows it delivered to settle (`Message.OnSettled`), then releases its pin (`mq.Releaser`). `Halt` on the claiming consumer ends the ticks and halts every unit at once; the worker calls it before its final flush, and the claims' stop releases the units and resigns the membership lease after it. A unit taken over from a slot that is no longer live is reset (`ResetOrphaned`) before it is bound — waiting, unbound, while the broker answers `ErrUnitHeld` (the dead owner's pin has not lapsed, or the unit was active within its pinned TTL), up to 30s — so the dead owner's unacked rows come back at once; on a process's first tick only if it is the one live member (a restart after a clean stop; after a crash the dead run's lease still counts as live for a lease duration). Each unit's rows delivered and unsettled are capped at its share of the worker's 10,000 (`ClaimConfig.MaxHeld`): an even share over the units assigned (`unitShare`, recomputed each tick and read by the broker before every fetch through `ConsumerConfig.MaxHeld`), never under 1,000, two batches; at its share a unit fetches only what keeps its pin, and one stuck unit never takes another's. A row stops counting at its first ack or nak attempt (`Message.OnSettled` fires whether or not the broker confirms), or after `ack_wait` if the worker never settles it, since the broker redelivers it then as a new row. A configured unit whose delivery ends, or whose durable or stream is gone or whose durable no longer fits (`mq.ErrConsumerNotFound`, `mq.ErrConsumerMismatch`) when it is bound, fails the worker; any other bind failure is retried next tick, logged as an error once it has lasted a minute, and an extra's end is logged. The lowest live slot counts the units with rows and no owner (`Sharded.Unowned`) for `wavehouse_ingest_shards_unowned`. @@ -286,7 +286,7 @@ Client POST /v1/ingest?table={table} error code (exception_code) and message — an unknown field, a MATERIALIZED/ALIAS column, or a column this role may not write is 117 (an EPHEMERAL value is accepted only where the format names columns, the - role may write it and a DEFAULT reads it; it feeds that DEFAULT, never stored or published); a record chtypes cannot answer for is a distinct "declined" outcome + role may write it, a DEFAULT reads it and no computed column does; it feeds that DEFAULT, never stored or published); a record chtypes cannot answer for is a distinct "declined" outcome (422), never a data rejection → Evaluate the role's check clauses over the accepted rows with one compiled chtypes filter — false is 403 for that record, unevaluable is 422 diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 79b44d26..b18f648f 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -144,7 +144,7 @@ Releases are built with [GoReleaser](https://goreleaser.com/). The configuration ### Supported Platforms -The binary requires cgo (dlopen only — no static link to the chtypes artifact) and glibc 2.34 or later, which sets the supported platform matrix: +The binary requires cgo (dlopen only — no static link to the chtypes artifact) and, on Linux, glibc 2.34 or later, which sets the supported platform matrix: | OS | Architecture | | -- | ----------- | @@ -645,7 +645,7 @@ For local development, `docker compose -f deployments/compose/dependencies.yaml WaveHouse uses a **Bring Your Own Schema** model. You create your tables in ClickHouse with whatever columns and engines you need. WaveHouse discovers the schemas automatically via `system.columns` and validates ingest data against them — see [Schema Validation](/api#post-v1ingesttabletable--ingest-data) for the rules a record must satisfy. -Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`/`ALIAS` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's implicit zero value where none is declared, evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). +Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's implicit zero value where none is declared, evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). Example table: diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index 83edc082..c52b728c 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -29,7 +29,7 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --frozen --lock chtypes.lock 26.8 ``` -It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–290 MB — expect the first run to take a minute or two. Without it the API process refuses to boot (`make dev`, `make test-e2e`), and the unit tests that need the engine skip; set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (CI does) to make a missing artifact fail those tests instead. A `503` on ingest, with row-filtered stream rows withheld, is what a process that did find an artifact answers for a ClickHouse line the artifact does not cover. +It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–290 MB — expect the first run to take a minute or two. Without it the API process refuses to boot (`make dev`, `make test-e2e`, and the app that `make test-integration` and `make ci` start), and the unit tests that need the engine skip; set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (CI does) to make a missing artifact fail those tests instead. A `503` on ingest, with row-filtered stream rows withheld, is what a process that did find an artifact answers for a ClickHouse line the artifact does not cover. ### Auto-installed by `make tools` @@ -351,11 +351,11 @@ Each test target writes `covdata` to `tmp/coverage//data/`, renders a tex | -------- | -------- | ------- | ------- | | Unit tests | `internal/*/_test.go` | No | `make test` | | SDK unit tests | `clients/ts/src/**/*.test.ts` | No | `make test-ts` (always includes coverage + gate) | -| Integration tests (Go) | `tests/integration/*_test.go`, `internal/cache/*_integration_test.go`, plus `internal/mq/natsspike` and `internal/mq`'s integration-tagged tests | Yes | `make test-integration` | +| Integration tests (Go) | `tests/integration/*_test.go`, `internal/cache/*_integration_test.go`, plus `internal/mq/natsspike`, and `internal/mq`'s and `internal/api`'s integration-tagged tests | Yes | `make test-integration` | | E2E tests (SDK) | `tests/e2e/sdk/*.test.ts` | Yes | `make test-e2e` | - **Unit tests** live beside the code they test (e.g., `internal/discovery/discovery_test.go`). They use mocks or embedded NATS (in-process, no Docker needed). -- **Integration tests** use the `//go:build integration` build tag. In `tests/integration`, `TestMain` starts one ClickHouse testcontainer and a dynamodb-local one (for the DynamoDB dedupe backend's tests), and boots the production wiring against it through `app.New` (embedded NATS, ingest worker, sweeper, hub, the API server on a random loopback port); tests reach it via `env(t)` and create their own tables. `TestNATSBackend_EndToEnd` also starts a NATS container configured from `deployments/nats/values.yaml`, applies `deployments/nats/jetstream.yaml` to it through `internal/mq/natstest`, and boots two processes on `mq.backend: nats` against it. `TestMain` also builds the `wavehouse` binary while the containers start, so the build is not charged to the `-timeout`: the `TestRoles_*` tests run it as separate OS processes (two API, two then three ingest, one killed with `SIGKILL`) over NATS, Redis and dynamodb-local, and read each ingest process's shard ownership from its `/metrics`. DLQ tests use `assert.Eventually` with a 30-second timeout for the 5-second ingest worker batch window. `internal/cache`'s integration tests start their own containers instead — Redis, Valkey, Dragonfly and a one-node Redis Cluster — for the shared backend. `shared_cache_test.go` starts its own Redis testcontainer per test (`startRedis`) and boots extra, independent `cache.backend: redis` instances over that same ClickHouse (`bootRedisApp`), to exercise the cache shared across processes rather than one package in isolation. The same target also runs `internal/mq/natsspike`. That package pins the nats-server behavior the external-NATS topology depends on, against an in-process server with no Docker. It lives under `internal/mq` because only that tree may import NATS, and it runs here rather than in the unit suite because each test takes seconds and the unit suite has a 15-second limit per package. For the same reason the external NATS broker's tests (`internal/mq/external*_test.go`, including its run of the `mqtest` conformance suite) and the NATS KV lease tests (`internal/mq/lease_test.go`, including their run of the `coordtest` conformance suite) carry the `integration` tag inside `internal/mq`, and the target runs them by name, so the package's untagged tests stay in the unit suite alone. +- **Integration tests** use the `//go:build integration` build tag. In `tests/integration`, `TestMain` starts one ClickHouse testcontainer and a dynamodb-local one (for the DynamoDB dedupe backend's tests), and boots the production wiring against it through `app.New` (embedded NATS, ingest worker, sweeper, hub, the API server on a random loopback port); tests reach it via `env(t)` and create their own tables. `TestNATSBackend_EndToEnd` also starts a NATS container configured from `deployments/nats/values.yaml`, applies `deployments/nats/jetstream.yaml` to it through `internal/mq/natstest`, and boots two processes on `mq.backend: nats` against it. `TestMain` also builds the `wavehouse` binary while the containers start, so the build is not charged to the `-timeout`: the `TestRoles_*` tests run it as separate OS processes (two API, two then three ingest, one killed with `SIGKILL`) over NATS, Redis and dynamodb-local, and read each ingest process's shard ownership from its `/metrics`. DLQ tests use `assert.Eventually` with a 30-second timeout for the 5-second ingest worker batch window. `internal/cache`'s integration tests start their own containers instead — Redis, Valkey, Dragonfly and a one-node Redis Cluster — for the shared backend. `shared_cache_test.go` starts its own Redis testcontainer per test (`startRedis`) and boots extra, independent `cache.backend: redis` instances over that same ClickHouse (`bootRedisApp`), to exercise the cache shared across processes rather than one package in isolation. The same target also runs `internal/mq/natsspike`. That package pins the nats-server behavior the external-NATS topology depends on, against an in-process server with no Docker. It lives under `internal/mq` because only that tree may import NATS, and it runs here rather than in the unit suite because each test takes seconds and the unit suite has a 15-second limit per package. For the same reason the external NATS broker's tests (`internal/mq/external*_test.go`, including its run of the `mqtest` conformance suite) and the NATS KV lease tests (`internal/mq/lease_test.go`, including their run of the `coordtest` conformance suite) carry the `integration` tag inside `internal/mq`, and the target runs them by name, so the package's untagged tests stay in the unit suite alone. `internal/api`'s `TestIntegration_*` tests (the read path's filters against a ClickHouse in a non-UTC zone) are run by name the same way. Shared test utilities live in `internal/testutil/`. The packages log through `slog.Default()`, so tests reach log output through `internal/testutil/logtest`: `logtest.Silence()` in a package's `TestMain` discards it, and `logtest.Capture(t, level)` routes it to a buffer for a test that asserts on log lines — such a test must not call `t.Parallel()`, because the default logger is process-wide. A test that needs the chtypes artifact takes its engine from `internal/typelayer/typelayertest` (`typelayertest.TestEngine`, and `typelayertest.SkipWithoutArtifact` to skip when the artifact is not installed — set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` to fail instead); `internal/typelayer`'s own tests use an in-package copy of that helper. A test that starts the embedded broker keeps its store in `internal/testutil/storedir`'s `storedir.New(t)` rather than a bare `t.TempDir()` (`testutil.NewEmbeddedMQ` does): the NATS server can finish writing a consumer's state after `Close` returns, which fails `t.TempDir`'s one-shot removal, and `storedir` removes the store again until those writes have landed ([#442](https://github.com/Wave-RF/WaveHouse/issues/442)). diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index b2811958..5abe056d 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -25,7 +25,7 @@ if (error?.code === 'ABORTED') { The SDK **never throws** for anything the server returns — all API errors come back in `Result.error`. It does throw on caller and environment errors: a non-absolute `baseURL` (REST calls reject with a `TypeError`; streams report `SSE_CONNECT_ERROR` to the subscriber's `error` callback — see [Serving under a path prefix](/sdk#serving-under-a-path-prefix)), `.stream()` / `.liveQuery()` in a runtime with no global `fetch` and no `options.fetch` (see [Runtime support](/sdk#runtime-support)), and an `auth` callback that rejects — a token-refresh failure propagates out of the REST call, and on a stream is reported as a retryable `SSE_AUTH_ERROR`. One more exception escapes an SDK call synchronously, though it is yours rather than ours: your own `status` handler throwing on the first `.subscribe()` or `.liveQuery()`, described under *If your own callback throws* below. -`code` and `retryable` are the server's own when its error body carries them — a failed ClickHouse query does, with codes like `clickhouse.rejected` and `clickhouse.unavailable` ([the full list](/api#clickhouse-errors-on-the-query-paths)). Otherwise `code` is `HTTP_` and a `5xx` is retryable. One exception to the table below: a [pipe that writes](/pipes#pipes-that-write) answers every ClickHouse failure `retryable: false` with no `Retry-After`, `clickhouse.unavailable` and `clickhouse.unknown` included, so the SDK returns it on the first attempt. +`code` and `retryable` are the server's own when its error body carries them — a failed ClickHouse query does, with codes like `clickhouse.rejected` and `clickhouse.unavailable` ([the full list](/api#clickhouse-errors-on-the-query-paths)). Otherwise `code` is `HTTP_` and a `5xx` is retryable. One exception to the table below: a [pipe that writes](/pipes#pipes-that-write) answers every ClickHouse failure `retryable: false` with no `Retry-After`, `clickhouse.unavailable` and `clickhouse.unknown` included, so the SDK returns it on the first attempt. An ingest `500` that carries `retryable: false` (a role whose insert permissions cannot be enforced on the table) is likewise returned on the first attempt. | Status | Code | Retryable | Description | |--------|------|-----------|-------------| @@ -35,7 +35,7 @@ The SDK **never throws** for anything the server returns — all API errors come | 404 | `HTTP_404` | No | Table, pipe, or tenant not found | | 400 | `clickhouse.rejected` / `clickhouse.limit_exceeded` | No | ClickHouse refused the query (bad SQL, an unknown column, a type mismatch) or it outran a limit — including the role's own caps | | 403 | `clickhouse.access_denied` | No | ClickHouse's user lacks a grant the statement needs | -| 500 | `HTTP_500` | Yes | Server error (retried per `maxRetries`) | +| 500 | `HTTP_500` | Yes, unless the body says `retryable: false` | Server error (retried per `maxRetries`) | | 500 / 502 | `clickhouse.unknown` | Yes | ClickHouse failed with no verdict (no exception code, no recognizable transport error); `502` on `wh.sql` | | 502 | `clickhouse.misconfigured` | No | ClickHouse refused WaveHouse's own credentials or database, a read ran under a `readonly=1` profile and was refused as a write, or the route to it is wrong (a redirect, or a `4xx` other than `408`/`413`/`429`, with no exception code) — an operator fix | | 502 | `clickhouse.response_too_large` | No | A response over the 64 MiB cap, on any query path (structured query, pipe or `wh.sql`) | @@ -147,7 +147,7 @@ Codegen reads `/v1/ops/schema`, which is **admin-only**. Against a non-dev serve | `--out`, `-o` | Output .d.ts file path | `./wavehouse.d.ts` | | `--auth`, `-a` | Bearer token (if auth required) | — | -The generated row type is the **read** shape, and computed columns are where it and the server disagree. An `EPHEMERAL` column declares a default, so codegen emits it, yet no query can ever return it — the type says readable where only the write is real. `MATERIALIZED` and `ALIAS` columns declare defaults too, so they are emitted as optional, but supplying either on `insert` is a `400` carrying ClickHouse's own code 117 (`Unknown field found while parsing JSONEachRow format: x`), and the type will not catch it. An `EPHEMERAL` value is accepted on a JSON `insert` when the role may write the column, a `DEFAULT` column reads it and no `MATERIALIZED` or `ALIAS` column does; it feeds that default and is never stored or returned. Otherwise it is a `400` with code 117, like an unknown column. Omit computed columns; the server fills them in. +The generated row type is the **read** shape, and computed columns are where it and the server disagree. An `EPHEMERAL` column declares a default, so codegen emits it, yet no query can ever return it — the type says readable where only the write is real. `MATERIALIZED` and `ALIAS` columns declare defaults too, so they are emitted as optional, but supplying either on `insert` is a `400` carrying ClickHouse's own code 117 (`Unknown field found while parsing JSONEachRow format: x`), and the type will not catch it. An `EPHEMERAL` value is accepted on a JSON `insert` when the role may write the column, a `DEFAULT` column reads it and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column does; it feeds that default and is never stored or returned. Otherwise it is a `400` with code 117, like an unknown column. Omit computed columns; the server fills them in. **Example output:** From 622101986503890cd659f5d0feac72083c324305 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 08:36:39 -0400 Subject: [PATCH 43/70] docs: tighten EPHEMERAL, timestamp rendering and check-literal claims Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- AGENTS.md | 4 ++-- CHANGELOG.md | 4 ++-- docs/src/content/docs/access-control.mdx | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index f88eb683..64dc5f81 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -47,7 +47,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes - **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row via `typelayer`, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns (a denied column stays as `MATERIALIZED` of its default, so naming it is code 117), plus a `DEFAULT ''` per `_eq` check column, and any `EPHEMERAL` column the role may write that a `DEFAULT` reads — which is how column policy and auto-inject are answered with no Go-side record inspection; a role shape's handle pool grows to `min(GOMAXPROCS, 4)` under load. A role schema that cannot compile is `500` with `retryable:false`. Test helpers live in `internal/typelayer/typelayertest` (`TestEngine`, `SkipWithoutArtifact`). `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) +- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns (a denied column stays as `MATERIALIZED` of its default, so naming it is code 117), plus a `DEFAULT ''` per `_eq` check column, and any `EPHEMERAL` column the role may write that a `DEFAULT` reads and no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` expression reads — which is how column policy and auto-inject are answered with no Go-side record inspection; a role shape's handle pool grows to `min(GOMAXPROCS, 4)` under load. A role schema that cannot compile is `500` with `retryable:false`. Test helpers live in `internal/typelayer/typelayertest` (`TestEngine`, `SkipWithoutArtifact`). `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) - **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -72,7 +72,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 16. **Bearer-token-only CORS posture (security)** — Bearer JWT on every request, no cookies/sessions; `corsMiddleware` deliberately **never** emits `Access-Control-Allow-Credentials` (not needed, and `*` + credentials is a spec violation browsers reject). `cors.allowed_origins` (settings directory, per tenant: a tenant route is decorated from the list of the tenant it names, everything else from tenant `0`'s — `corsOrigins`) controls who can *read* responses, not cookie scope; CSRF protection is structural. Don't reintroduce cookie auth or `Allow-Credentials` without a design discussion — answers GitHub #29/#30. Code: `internal/api/router.go`. 17. **Non-fatal boot** — schema-discovery failure on boot is non-fatal: `internal/app` records an `api.BootState`, binds `:8080`, serves 503 on `/livez`/`/readyz` with the diagnostic, and retries via `SchemaRegistry.RetryRefresh` (jittered backoff 2s → 60s), per tenant over a nested directory: `/livez` is 503 while no tenant has completed a first discovery, then sticky 200, and a tenant's outage after that is its log line and counter, never a probe failure. Until a tenant's first discovery its table lookups are a 503 with `Retry-After`, not a 404. Bounds supervisor restart loops. 18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention; `/readyz` pings every open ClickHouse pool at once and is ready at the first answer, 503 naming each when none answers); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. -19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), as RFC 3339 in UTC at the column's scale (`date_time_output_format=iso`, e.g. `"2026-06-21T04:00:00.123Z"`), the spelling `/v1/query` and pipes pin too — there is no separate WaveHouse rewrite step to keep in sync, so live and query reads can't drift on spelling *or* instant (#372), whatever zone the column or server uses. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. +19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), as RFC 3339 in UTC at the column's scale (e.g. `"2026-06-21T04:00:00.123Z"`), because `date_time_output_format=iso` is pinned on both surfaces — the ingest export settings (`internal/typelayer/ingest.go`) and the read path's `chReadSettingsFixed` (`internal/api/clickhouse_http.go`), which must change together with a `chRendering` bump — so there is no separate WaveHouse rewrite step to keep in sync and live and query reads can't drift on spelling *or* instant (#372). The column's or server's zone affects only how zone-less *input* is read, never the rendered spelling. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. 20. **Sealed MQ boundary** — only `internal/mq` imports NATS/JetStream (`github.com/nats-io/…`), enforced by the `depguard` rule in `.golangci.yml`, so `make lint` fails on a leak in every package it builds (the `integration`-tagged files under `tests/` are outside lint's build context — keep them clean by convention, through `mq.Broker`). A test outside `internal/mq` that needs a real NATS server goes through `internal/mq/natstest`, which stands one up from the shipped `deployments/nats` files and hands back a URL and passwords, never a NATS type. The boundary is semantic as well: everything else addresses events by `mq.Topic` and states intent through mq-owned interfaces (`Publisher`, `Consumer`, `DeadLetterer`, `Purger`, `Replayer`, …), and never builds a subject, names a stream, or reasons in sequences — so a subject, stream, or broker change lands in one package ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 4; story 5's tenant token landed there alone — `Topic.Tenant`, first in every subject). Don't add a raw accessor (`JetStream()`, `NatsConn()`, `GetServer()`) back, and don't hand-build `"ingest."`/`"dlq."` subjects outside `internal/mq` — widen the mq surface with an intent-level method instead. 21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503` (generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds every row of a role that has a row `filter` with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and, on Linux, glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. diff --git a/CHANGELOG.md b/CHANGELOG.md index f1af194d..8866f1fc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -84,9 +84,9 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **A role's ingest handle pool grows under load** (`internal/typelayer/pool.go`): a role shape's pool of compiled handles starts at one and grows to `min(GOMAXPROCS, 4)` when every handle is busy, as a base table's grows to `min(GOMAXPROCS, 8)`, so concurrent inserts by one role no longer serialize on a single handle. - **Client-side timestamp comparison in the SDK is by instant** (`clients/ts/src/timestamp.ts` (new), `clients/ts/src/{query-builder,stream/live-query}.ts`): two strings that both read as timestamps are compared to the nanosecond for `=`, `!=`, `in`, `>`, `>=`, `<`, `<=` in stream `where` filters, with a zone-less value read as UTC; the live-query backfill seam compares at the coarser of the two precisions, so an event in the boundary row's own millisecond is covered. Other strings keep strict equality and string order. - **Test helpers moved to `internal/typelayer/typelayertest`** (internal): `TestEngine`, `SkipWithoutArtifact` and `RequireEnv` no longer live in `internal/typelayer`, which now never imports `testing`. -- **A policy `check` on a column the table cannot accept is now refused instead of silently unenforced** (BREAKING; `internal/api/ingest.go`, `internal/discovery/discovery.go`, `docs/src/content/docs/{api.md,access-control.mdx}`): a `check` clause naming a column the table does not have, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one can never be enforced — the published row carries one slot per insertable column, so an auto-injected value for anything outside that set is dropped on the way out, and an ephemeral column is never stored even though the row does carry it. The record inserted **without** the value the policy required and answered `200 {"ok":true}`. Demonstrated on this branch: a `check` of `tenant _eq {{ jwt.tenant }}` against a `MATERIALIZED tenant` column published `columns:["page"], row:["/a"]` — the tenant constraint absent from the row entirely. It is now refused, naming every offending column and the reason, on **every** insert by that role until the policy or the table is corrected: a single-object request answers `403`, while a batch answers `200` with the same message against each record in `results` — the batch is still read to the end and reports per record, as it does for any other rejection. Policy validation cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**; `wavehouse validate` will not tell you. +- **A policy `check` on a column the table cannot accept is now refused instead of silently unenforced** (BREAKING; `internal/api/ingest.go`, `internal/discovery/discovery.go`, `docs/src/content/docs/{api.md,access-control.mdx}`): a `check` clause naming a column the table does not have, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one can never be enforced — the published row carries one slot per insertable column, so an auto-injected value for anything outside that set is dropped on the way out, and an ephemeral column is never stored even though the row does carry it. The record inserted **without** the value the policy required and answered `200 {"ok":true}`. Demonstrated on this branch: a `check` of `tenant _eq {{ jwt.tenant }}` against a `MATERIALIZED tenant` column published `columns:["page"], row:["/a"]` — the tenant constraint absent from the row entirely. It is now refused, naming every offending column and the reason, on **every** insert by that role until the policy or the table is corrected: a single-object request answers `403`, while a batch answers `200` with the same message against each record in `results` — the batch is still read to the end and reports per record, as it does for any other rejection. Policy validation cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**; `wavehouse validate` will not tell you. Superseded in part: on this branch the published row never carries an ephemeral value (see the `EPHEMERAL` entry above). -- **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` (the `InsertableColumns` / `InsertableColumnNames` lists it first added are removed again, an internal API: which columns a record may supply is the type layer's `Table.WireColumns`). `EPHEMERAL` columns are likewise left out of the envelope, and this entry's earlier claim that they stay insertable is superseded by the `EPHEMERAL` entry above. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. +- **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` (the `InsertableColumns` / `InsertableColumnNames` lists it first added are removed again, an internal API: which columns a record may supply is the type layer's `Table.WireColumns`). `EPHEMERAL` columns are likewise left out of the envelope, and this entry's earlier claim that they stay insertable is superseded by the `EPHEMERAL` entry above. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. Superseded: that rejection is now ClickHouse's own `400` with `exception_code` 117, `Unknown field found while parsing …` (see the type-layer entry). - **Ingest reads the request body up front, and the per-record decisions sit behind interfaces** (`internal/api/{ingest,ingest_seams,bufpool,record_reader}.go`, `internal/stream/hub.go`, `internal/ingest/compact.go`): responses are unchanged except at the body cap and one new read-failure body (`400 {"error":"invalid request body"}`, when the body cannot be read at all — a malformed transfer encoding or a truncated upload, which previously surfaced through the decoder as `invalid json`), and at the cap the `413` is now decided before any record is processed: an over-cap batch no longer ingests the prefix it had already decoded, and an over-cap single-object body whose first object was followed by an oversized tail — which used to answer `200` after ingesting that one object — now answers `413`. Both are improvements, since a client retrying a `413` can no longer double-insert a prefix, but they are behavior changes and the memory profile changes too (see below); this is the seam work the native type layer lands against. The handler now reads the whole (already `MaxBytesReader`-capped) body into a pooled `*bytes.Buffer` and runs the record readers over those bytes rather than the live connection — so the `413` surfaces at that read instead of mid-iteration (same status, same message), and the `415` is decided from the header before a single byte is read. Three decision points became interfaces with default implementations that delegate to today's code unchanged: `RecordValidator` (schema validation + timestamp canonicalization — the two calls stay where they are, with the check-clause block between them, since merging them would move checks onto canonicalized values), `InsertChecker` (the `_eq` and `_in` comparisons), and `stream.RowEvaluator` (row visibility, reached by both the live fan-out and replay through the one shared admission step). **The memory profile is not unchanged, and that is the deliberate part.** Streaming meant peak resident bytes on the order of one record: the NDJSON path scanned line by line and the array path let `json.Decoder` compact after each element. Peak is now O(body) per in-flight request — and `bytes.Buffer` grows by doubling, so the peak allocation can exceed the body cap before `MaxBytesReader` errors. `maxPooledBufferBytes` (1 MiB) caps what a request hands *back* to the pool, not its peak, and nothing in `internal/api` bounds total in-flight bytes, so the ceiling is concurrency × the 16 MiB data-plane cap — which has no operator knob (`maxRequestBytes` is test-only), so the outer limit is the reverse proxy's, which the reverse-proxy guide already advises setting. Kept because it is the shape the native type layer lands against, which needs the body addressable rather than consumed; a bound on total in-flight ingest bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). Operators fronting large batches at high concurrency should size for it or cap body size at the proxy. All three are nil-safe: an un-wired handler or `Hub` uses the default rather than panicking past the check. Also new: `ingest.EncodeCompactRow`, which renders a record as one `JSONCompactEachRow` line — inert in this commit, and the encoder every published row went through at that point in the release (superseded: the published row is now ClickHouse's own export, and `EncodeCompactRow`, `RecordValidator` and `InsertChecker` are gone — see the type-layer entry). diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 879bdb6a..1c1005ce 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -252,7 +252,7 @@ Compare claims against `String` or `UUID` columns (tenant ids, org ids, user ids On an integer column (`UInt8` through `UInt256`, `Int8` through `Int256`, `Nullable` included) the claim must still fit the column's type, and is compared through a strict cast: a claim that is not the canonical spelling of a value the column can hold — out of range such as `18446744073709551616` (2^64), or spelled `007` or `+5` — matches no rows on any operator, and an insert `check` refuses the record with `403`. It never wraps onto another value. -On a timestamp column (`Date`, `DateTime`, `DateTime64`) the claim is compared the way ClickHouse compares a string with that type. Write it as `YYYY-MM-DD hh:mm:ss[.fff]` with no offset: it is read in the column's time zone, or the server's when the column declares none. A claim in RFC 3339 form such as `2026-01-01T00:00:00Z` is refused on a `DateTime` column (ClickHouse code 53, measured on 24.8 and 26.8) rather than matched, so for timestamps use the zone-less form, or compare claims against a String or UUID column instead. +On a timestamp column (`Date`, `DateTime`, `DateTime64`) the claim is compared the way ClickHouse compares a string with that type. Write it as `YYYY-MM-DD hh:mm:ss[.fff]` with no offset: it is read in the column's time zone, or the server's when the column declares none. On the read path, where the claim is bound as a plain `String`, a claim in RFC 3339 form such as `2026-01-01T00:00:00Z` is refused on a `DateTime` column (ClickHouse code 53, measured on 24.8 and 26.8) rather than matched. An insert `check` literal is parsed under `best_effort`, so an offset spelling such as `2026-06-21T06:00:00+02:00` is accepted there and stored as its instant. For a `filter`, use the zone-less form, or compare claims against a String or UUID column instead. Prefer identity columns over time columns, and express time windows in the query instead. ::: From dc9ee8ff8394d9ea9a3e6ca5a7f3b98c5ddf0a45 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 09:04:59 -0400 Subject: [PATCH 44/70] chore: label and coverage-exclude the typelayer package Map internal/typelayer/**, chtypes.lock and scripts/fetch-chtypes.sh to area/typelayer in the PR labeler, and exclude the test-only internal/typelayer/typelayertest from coverage like coordtest, mqtest and natstest. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018uEnYtmudjD1nn3T44zuhB --- .github/labeler.yml | 7 +++++++ .testcoverage.yml | 3 +++ CHANGELOG.md | 2 +- 3 files changed, 11 insertions(+), 1 deletion(-) diff --git a/.github/labeler.yml b/.github/labeler.yml index 493b7555..d6cd4ae9 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -54,6 +54,13 @@ - any-glob-to-any-file: - "internal/pipes/**" +"area/typelayer": + - changed-files: + - any-glob-to-any-file: + - "internal/typelayer/**" + - "chtypes.lock" + - "scripts/fetch-chtypes.sh" + "area/sdk": - changed-files: - any-glob-to-any-file: diff --git a/.testcoverage.yml b/.testcoverage.yml index 744849d4..87e8837f 100644 --- a/.testcoverage.yml +++ b/.testcoverage.yml @@ -57,6 +57,9 @@ exclude: # internal/mq/natstest/ stands up NATS as an operator deploys it, for # tests outside internal/mq; test code, like mqtest. - ^internal/mq/natstest/ + # internal/typelayer/typelayertest/ opens a test Engine on the locked + # chtypes artifact for other packages' tests; test code, like mqtest. + - ^internal/typelayer/typelayertest/ - ^tests/ # scripts/ holds Go helpers (cov, orchestrator) that drive the build but # aren't part of the shipped binary; they show up in `-coverpkg=./...` diff --git a/CHANGELOG.md b/CHANGELOG.md index 8866f1fc..10b88f86 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -83,7 +83,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **A JSON array body with content after its closing `]` is refused** (`internal/api/ingest_framing.go`): the request answers `400 {"error":"invalid json: content after the closing ']' of the json array"}` and publishes nothing. - **A role's ingest handle pool grows under load** (`internal/typelayer/pool.go`): a role shape's pool of compiled handles starts at one and grows to `min(GOMAXPROCS, 4)` when every handle is busy, as a base table's grows to `min(GOMAXPROCS, 8)`, so concurrent inserts by one role no longer serialize on a single handle. - **Client-side timestamp comparison in the SDK is by instant** (`clients/ts/src/timestamp.ts` (new), `clients/ts/src/{query-builder,stream/live-query}.ts`): two strings that both read as timestamps are compared to the nanosecond for `=`, `!=`, `in`, `>`, `>=`, `<`, `<=` in stream `where` filters, with a zone-less value read as UTC; the live-query backfill seam compares at the coarser of the two precisions, so an event in the boundary row's own millisecond is covered. Other strings keep strict equality and string order. -- **Test helpers moved to `internal/typelayer/typelayertest`** (internal): `TestEngine`, `SkipWithoutArtifact` and `RequireEnv` no longer live in `internal/typelayer`, which now never imports `testing`. +- **Test helpers moved to `internal/typelayer/typelayertest`** (internal; `.testcoverage.yml`, `.github/labeler.yml`): `TestEngine`, `SkipWithoutArtifact` and `RequireEnv` no longer live in `internal/typelayer`, which now never imports `testing`. The helper package is excluded from coverage like the other test-only packages, and a PR touching `internal/typelayer/`, `chtypes.lock` or `scripts/fetch-chtypes.sh` is labeled `area/typelayer`. - **A policy `check` on a column the table cannot accept is now refused instead of silently unenforced** (BREAKING; `internal/api/ingest.go`, `internal/discovery/discovery.go`, `docs/src/content/docs/{api.md,access-control.mdx}`): a `check` clause naming a column the table does not have, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one can never be enforced — the published row carries one slot per insertable column, so an auto-injected value for anything outside that set is dropped on the way out, and an ephemeral column is never stored even though the row does carry it. The record inserted **without** the value the policy required and answered `200 {"ok":true}`. Demonstrated on this branch: a `check` of `tenant _eq {{ jwt.tenant }}` against a `MATERIALIZED tenant` column published `columns:["page"], row:["/a"]` — the tenant constraint absent from the row entirely. It is now refused, naming every offending column and the reason, on **every** insert by that role until the policy or the table is corrected: a single-object request answers `403`, while a batch answers `200` with the same message against each record in `results` — the batch is still read to the end and reports per record, as it does for any other rejection. Policy validation cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**; `wavehouse validate` will not tell you. Superseded in part: on this branch the published row never carries an ephemeral value (see the `EPHEMERAL` entry above). - **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` (the `InsertableColumns` / `InsertableColumnNames` lists it first added are removed again, an internal API: which columns a record may supply is the type layer's `Table.WireColumns`). `EPHEMERAL` columns are likewise left out of the envelope, and this entry's earlier claim that they stay insertable is superseded by the `EPHEMERAL` entry above. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. Superseded: that rejection is now ClickHouse's own `400` with `exception_code` 117, `Unknown field found while parsing …` (see the type-layer entry). From eab6e3dcf5965d66398f382c88b7993c9211a1a5 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 09:20:22 -0400 Subject: [PATCH 45/70] chore: correct comments the type-layer migration left stale Comments that still described deleted mechanisms: precomputed timestamp specs, the native driver serving structured queries and pipes, the type layer padding every verdict list to the body's line count, clickhouse-go binding the builder's positional placeholders, the Go-side insert check feeding client literals to CanonicalScalar, and an omitted column riding as an explicit null. Also scopes "an explicit null takes the default" to non-Nullable columns, describes a denied insert column as re-declared MATERIALIZED, and words the plaintext-HTTP-hop warning as covering every query and insert. Co-Authored-By: Claude Opus 5.5 --- clients/ts/src/types.ts | 15 ++++----- docs/src/components/LiveDemo.astro | 12 ++++--- internal/api/ingest_framing.go | 9 +++--- internal/api/ingest_test.go | 10 +++--- internal/api/router_test.go | 6 ++-- internal/chconn/chconn.go | 5 +-- internal/discovery/discovery.go | 17 +++++----- internal/ingest/types.go | 11 ++++--- internal/policy/canonical.go | 33 ++++++++++---------- internal/policy/policy.go | 6 ++-- internal/policy/policy_test.go | 24 +++++++------- internal/settings/settings.go | 18 ++++++----- internal/settings/validate.go | 8 ++--- internal/stream/hub_test.go | 19 +++++------ internal/typelayer/ingest.go | 11 ++++--- tests/e2e/sdk/ingest.test.ts | 14 +++++---- tests/integration/discovery_metadata_test.go | 3 +- 17 files changed, 116 insertions(+), 105 deletions(-) diff --git a/clients/ts/src/types.ts b/clients/ts/src/types.ts index bfd88ba9..1b82c924 100644 --- a/clients/ts/src/types.ts +++ b/clients/ts/src/types.ts @@ -307,18 +307,19 @@ export interface InsertRecordResult { error?: string; /** * ClickHouse's own numeric error code, present only when the server's parser - * is what refused the record — 117 unknown field, 27 unparseable value, 6 out - * of range. Absent for a gateway rejection (a failed policy check, a missing + * is what refused the record — 117 unknown field, 27 or 6 a value the column + * cannot read. Absent for a gateway rejection (a failed policy check, a missing * dedupe id), so `exception_code !== undefined` means "ClickHouse answered". * The same name carries it on a whole-request error body, beside the string * `code` class (reachable as `error.details`). * * 117 also covers **a column the caller's role may not write**. Column policy - * is enforced by compiling the role's own schema without the denied columns, - * so naming one is an unknown field to the parser rather than a separate - * gateway refusal: a `400` with this code, where it used to be a - * `403 column "x" not allowed for insert`. The message is ClickHouse's own and - * does not reveal whether the column exists. + * is enforced by compiling the role's own schema, where a denied column is + * re-declared as computed (`MATERIALIZED` of its default), so naming one is an + * unknown field to the parser rather than a separate gateway refusal: a `400` + * with this code, where it used to be a `403 column "x" not allowed for + * insert`. The message is ClickHouse's own and does not reveal whether the + * column exists. */ exception_code?: number; } diff --git a/docs/src/components/LiveDemo.astro b/docs/src/components/LiveDemo.astro index f2814a32..49a3095f 100644 --- a/docs/src/components/LiveDemo.astro +++ b/docs/src/components/LiveDemo.astro @@ -178,8 +178,9 @@ const DEMO_HOST = "stats.wavehouse.dev"; const posthog = () => (window as { posthog?: { capture: (event: string, props?: object) => void } }).posthog; - // Backfill rows come back RFC3339 from ClickHouse, but live SSE rows carry - // the producer's payload as ingested — a zone-less ClickHouse-style + // Backfill rows come back RFC3339 from ClickHouse. Live SSE rows do too from + // a server with the chtypes type layer, but an older one sends the producer's + // payload as ingested — a zone-less ClickHouse-style // "YYYY-MM-DD HH:MM:SS[.fff]" (UTC by the gh_events contract), which // Date.parse would read as LOCAL time. Normalize to ISO-8601 UTC. function normTs(ts: string): string { @@ -358,9 +359,10 @@ const DEMO_HOST = "stats.wavehouse.dev"; // --- Feed. const seen = new Set(); // Normalize every key part across the two row sources: backfill rows are - // ClickHouse-materialized (RFC3339 ts, ""/0 defaults) while live SSE rows - // are raw producer payloads (zone-less ts, fields possibly absent) — raw - // interpolation would never dedup a row seen via both paths. + // ClickHouse-materialized (RFC3339 ts, ""/0 defaults) while an older + // server's live SSE rows are raw producer payloads (zone-less ts, fields + // possibly absent) — raw interpolation would never dedup a row seen via + // both paths. const rowKey = (r: GhRow) => `${r.event_type}|${r.action || ""}|${r.actor_login}|${normTs(r.event_ts || "")}|${Number(r.number) || 0}`; diff --git a/internal/api/ingest_framing.go b/internal/api/ingest_framing.go index 338860c5..fb55a4b3 100644 --- a/internal/api/ingest_framing.go +++ b/internal/api/ingest_framing.go @@ -35,10 +35,11 @@ import ( // closing bracket shares that record's line and the reader cannot resync // past it. JSONEachRow needs no brackets, so removing them costs nothing and // makes every position salvageable, first and last included; -// - every other newline outside a string becomes a space. Not cosmetic: -// typelayer.Ingest pads its verdict list out to the body's newline count, so -// a pretty-printed array would come back with one phantom "no verdict" -// record per line of layout. This leaves exactly elements-1 newlines. +// - every other newline outside a string becomes a space. Not cosmetic: when +// chtypes declines a whole batch without per-record detail, the type layer +// counts the body's lines to answer each record, so a pretty-printed array +// would come back with one phantom declined record per line of layout. +// This leaves exactly elements-1 newlines, so that count stays right. // // A raw newline inside a string is illegal JSON, so leaving those alone costs // nothing and keeps the caller's bytes the caller's. diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index 46843f5c..55c0c3d8 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -768,9 +768,9 @@ func TestIngest_Policy_CheckClause_NullValue_StoredValueIsWhatIsChecked(t *testi // With a claim that DOES resolve, an explicit null takes the INJECTED value // rather than the table's own default: null_as_default resolves it against - // the ROLE's compiled schema, whose DEFAULT is the claim. A null on a checked - // column therefore behaves exactly like omitting it, and can never carry - // another tenant's value — the property that matters. + // the ROLE's compiled schema, whose DEFAULT is the claim. A null on a + // non-Nullable checked column therefore behaves exactly like omitting it, and + // can never carry another tenant's value — the property that matters. pub2 := &testutil.MockPublisher{} h2 := newTestIngestHandler(t, testRegistry(t), pub2) h2.PolicySource = h.PolicySource @@ -2600,8 +2600,8 @@ func publishedData(t *testing.T, pub *testutil.MockPublisher) map[string]any { } // publishedRow decodes one published envelope and zips its row by column name. -// A column the record omitted rides as an explicit null, exactly as it does on -// the wire, so a caller can tell "absent" from "present and null" only by value. +// A column the record omitted rides with the value ClickHouse filled in (its +// DEFAULT, or the type's default), exactly as it does on the wire. func publishedRow(t *testing.T, payload []byte) map[string]any { t.Helper() var evt ingest.EventMessage diff --git a/internal/api/router_test.go b/internal/api/router_test.go index 742c084c..d68782b6 100644 --- a/internal/api/router_test.go +++ b/internal/api/router_test.go @@ -539,9 +539,9 @@ func TestNewRouter_RawSQLAdminGate(t *testing.T) { t.Run("admin reaches handler", func(t *testing.T) { t.Parallel() - // nil driver.Conn would panic inside executeQuery, but the handler - // returns 400 before that on a missing body — which is enough to - // confirm the gate let the request through. + // A zero QueryHandler has no target getter and would panic past the + // body check, but the handler returns 400 before that on a missing + // body — which is enough to confirm the gate let the request through. rec := post("admin") assert.NotEqual(t, http.StatusNotFound, rec.Code, "admin must reach the handler") assert.NotEqual(t, http.StatusForbidden, rec.Code, "admin must not be 403'd") diff --git a/internal/chconn/chconn.go b/internal/chconn/chconn.go index b53949f2..4344d77c 100644 --- a/internal/chconn/chconn.go +++ b/internal/chconn/chconn.go @@ -6,8 +6,9 @@ // implements driver.Conn by delegating every call to the connection current // at that instant, so a consumer resolves its tenant's Manager per call and // never learns a resize happened; the HTTP-side consumers (ingest INSERTs, -// the raw-SQL proxy) read their tenant's Target per request and take their -// client from an HTTPClients. +// structured queries, pipes, the raw-SQL proxy) read their tenant's Target +// per request, and the worker and the proxy take their client from an +// HTTPClients. package chconn import ( diff --git a/internal/discovery/discovery.go b/internal/discovery/discovery.go index 4234f842..32fbb4d6 100644 --- a/internal/discovery/discovery.go +++ b/internal/discovery/discovery.go @@ -125,8 +125,8 @@ func columnNames(cols []Column) []string { // Lookup returns the named column and whether the table declares it. Matching // is exact, as ClickHouse's own column resolution is. Linear over Columns, which -// is the right shape for the per-record call sites: schemas are small and the -// caller asks about one or two columns. +// is the right shape for its call sites: schemas are small and the caller asks +// about one or two columns. // // It returns the Column rather than a bool because "does the table have it" is // rarely the whole question — a caller on the ingest path also has to know @@ -231,13 +231,12 @@ func (sr *SchemaRegistry) OnRefresh(hook RefreshHook) { sr.onRefresh = append(sr.onRefresh, hook) } -// Refresh rebuilds the in-memory schema cache: it discovers the server's default -// time zone and version, queries system.columns, attaches each table's DDL from -// system.tables, precomputes timestamp column specs, and then runs the -// OnRefresh hooks before marking the registry loaded. A refresh that started -// before the one whose snapshot is already published returns nil without -// publishing: the published refresh started later, so it saw everything -// committed before this one started. +// Refresh rebuilds the in-memory schema cache: it discovers the server's +// default time zone and version, queries system.columns, attaches each table's +// DDL from system.tables, and then runs the OnRefresh hooks before marking the +// registry loaded. A refresh that started before the one whose snapshot is +// already published returns nil without publishing: the published refresh +// started later, so it saw everything committed before this one started. func (sr *SchemaRegistry) Refresh(ctx context.Context) error { tracer := otel.GetTracerProvider().Tracer("wavehouse-discovery") ctx, span := tracer.Start(ctx, "SchemaRegistry.Refresh") diff --git a/internal/ingest/types.go b/internal/ingest/types.go index 8cdc23b2..7a8fd2a0 100644 --- a/internal/ingest/types.go +++ b/internal/ingest/types.go @@ -15,11 +15,12 @@ const FormatJSONCompactEachRow = "JSONCompactEachRow" // EventMessage is the wire format published to the MQ. // // Row data travels POSITIONALLY: Row is one JSONCompactEachRow line (a JSON -// array, no trailing newline) and Columns names its positions in the table's -// declaration order. The two are only meaningful together — a reader that -// cannot pair them (a length mismatch, an undecodable row, a repeated column -// name) has no way to map a value to a column and must fail closed rather than -// guess. +// array, no trailing newline) as ClickHouse's own writer produced it, and +// Columns names its positions — the table's wire columns, or the narrower list +// a column-restricted role writes, in declaration order. The two are only +// meaningful together — a reader that cannot pair them (a length mismatch, an +// undecodable row, a repeated column name) has no way to map a value to a +// column and must fail closed rather than guess. type EventMessage struct { TableName string `json:"table_name"` Scope string `json:"scope"` diff --git a/internal/policy/canonical.go b/internal/policy/canonical.go index b3714ef9..91b78b5b 100644 --- a/internal/policy/canonical.go +++ b/internal/policy/canonical.go @@ -18,10 +18,10 @@ import ( // maxCanonicalDigits bounds both a numeric literal's digit count and its // exact decimal expansion. Big-integer parsing is superlinear in digit count -// and the ingest check path hands CanonicalScalar client-controlled literals -// (CWE-400), and a short exponent literal can hide a wide expansion ("1e-150" -// is six characters with a 152-character exact form). 100 digits is far past -// any real id — uint256 is 78. +// and a claim's value is whatever the token carries (CWE-400), and a short +// exponent literal can hide a wide expansion ("1e-150" is six characters with +// a 152-character exact form). 100 digits is far past any real id — uint256 +// is 78. const maxCanonicalDigits = 100 // CanonicalScalar renders a decoded JSON value as the canonical string the @@ -32,25 +32,24 @@ const maxCanonicalDigits = 100 // fmt.Sprint's "map[…]"/"[…]" rendering would let _neq/_lt match essentially // every row; the one legitimate structured shape, a bare-claim _in array, is // unpacked by resolveInValues before its elements reach here. A json.Number -// (jwt.WithJSONNumber on claims, UseNumber on ingest payloads) binds in -// canonical decimal form, not the token's spelling: "1", "1.0", and "1e3" are -// one JSON value, and a numeric ClickHouse column rejects '1.0'/'1e3' as a -// per-query TYPE_MISMATCH error. The canonical form is exact at every width -// and precision — integer literals via big.Int, fractions and exponents via -// canonicalDecimal, never a float64 round-trip that could bind a value the -// token doesn't carry ("1e-400" fails closed rather than collapsing to "0"). -// A literal, or an exact form, past maxCanonicalDigits likewise has no -// canonical form and fails closed (1e400, 1e-400). Claim resolution and the -// insert-check comparison's two sides (internal/api) all route through this -// one function, so what a read filter binds and what a write check accepts +// (jwt.WithJSONNumber on claims) binds in canonical decimal form, not the +// token's spelling: "1", "1.0", and "1e3" are one JSON value, and a numeric +// ClickHouse column rejects '1.0'/'1e3' as a per-query TYPE_MISMATCH error. The +// canonical form is exact at every width and precision — integer literals via +// big.Int, fractions and exponents via canonicalDecimal, never a float64 +// round-trip that could bind a value the token doesn't carry ("1e-400" fails +// closed rather than collapsing to "0"). A literal, or an exact form, past +// maxCanonicalDigits likewise has no canonical form and fails closed (1e400, +// 1e-400). Claim resolution routes every filter and check value through this +// one function, so what a read filter binds and what a write check requires // can't drift. func CanonicalScalar(v any) (string, bool) { switch val := v.(type) { case nil, map[string]any, []any: return "", false case json.Number: - // The digit bound guards the superlinear big.Int parse on the - // client-controlled ingest path. It counts digits, not bytes — a sign, + // The digit bound guards the superlinear big.Int parse of a + // token-supplied value. It counts digits, not bytes — a sign, // point, or exponent marker doesn't feed the big parse — so "-1e99" // binds exactly like its written-out form, matching the "one JSON // value" contract below (only at the bound's very edge can the exact-form diff --git a/internal/policy/policy.go b/internal/policy/policy.go index 046a9304..f01fd82e 100644 --- a/internal/policy/policy.go +++ b/internal/policy/policy.go @@ -253,7 +253,7 @@ func evaluateSelect(perms *SelectPermissions, claims map[string]any) *ResolvedPe } // Resolve filters into WHERE clause. A bind-unsafe filter column can't be - // emitted safely — a '?' in it would shift clickhouse-go's positional value + // emitted safely — a '?' in it would shift the builder's positional `?` // binding, including this RLS filter's own bound value — so deny the role // fail-closed rather than drop the predicate (which would widen row access) // or emit a mis-bound query. validateSelectPerms rejects such a policy at @@ -860,8 +860,8 @@ func validateSelectPerms(table, role string, perms *SelectPermissions) error { return fmt.Errorf("table %q, op %q, role %q: max_memory_usage must be non-negative", table, op, role) } // Filter column names are interpolated into SQL (backtick-quoted) at query - // time, so a '?' in one would shift clickhouse-go's positional value - // binding. Refuse such a policy at write time, mirroring the query builder's + // time, so a '?' in one would shift the builder's positional `?` binding. + // Refuse such a policy at write time, mirroring the query builder's // chsql.BindUnsafe guard on caller-supplied columns. for col, f := range perms.Filter { if chsql.BindUnsafe(col) { diff --git a/internal/policy/policy_test.go b/internal/policy/policy_test.go index 2f0d851d..c7d5a677 100644 --- a/internal/policy/policy_test.go +++ b/internal/policy/policy_test.go @@ -480,12 +480,12 @@ func TestResolveTemplate_MultipleTemplates(t *testing.T) { } // TestCanonicalScalar pins the one rule every bound value flows through — claim -// templates, _in elements, and the ingest check comparison alike. The canonical -// form is exact at every width and precision (integers via big.Int, fractions -// and exponents via canonicalDecimal — never a float64 round-trip, which would -// collapse "1e-400" to "0" and round wide decimals onto their neighbors), and -// a literal or exact form past the 100-digit bound fails closed in both -// directions. +// templates and _in elements, for row filters and insert checks alike. The +// canonical form is exact at every width and precision (integers via big.Int, +// fractions and exponents via canonicalDecimal — never a float64 round-trip, +// which would collapse "1e-400" to "0" and round wide decimals onto their +// neighbors), and a literal or exact form past the 100-digit bound fails closed +// in both directions. func TestCanonicalScalar(t *testing.T) { t.Parallel() tests := []struct { @@ -525,10 +525,10 @@ func TestCanonicalScalar(t *testing.T) { // allocating a gigabyte-scale expansion before the length check runs. {"zero mantissa with out-of-range exponent", json.Number("0e201"), "", false}, // The 100-digit literal bound: big.Int work is superlinear in digit - // count and the ingest path hands this function client-controlled - // literals, so anything longer fails closed before any parsing. The - // bound counts digits, not bytes — sign and exponent markers ride free — - // so two spellings of one value pass or fail together. + // count and a claim's value is whatever the token carries, so anything + // longer fails closed before any parsing. The bound counts digits, not + // bytes — sign and exponent markers ride free — so two spellings of one + // value pass or fail together. {"100-digit integer at the bound stays exact", json.Number(strings.Repeat("9", 100)), strings.Repeat("9", 100), true}, {"101-digit literal has no canonical form", json.Number(strings.Repeat("9", 101)), "", false}, {"digit bound ignores sign and exponent bytes", json.Number("-1e99"), "-1" + strings.Repeat("0", 99), true}, @@ -978,8 +978,8 @@ func TestEvaluate_FilterUnresolvableClaim_FailsClosed(t *testing.T) { } // TestValidate_RejectsBindUnsafeFilterColumn: a policy whose row-filter column -// contains '?' is refused at write time — it would shift clickhouse-go's -// positional value binding when interpolated into the WHERE clause. +// contains '?' is refused at write time — it would shift the builder's +// positional `?` binding when interpolated into the WHERE clause. func TestValidate_RejectsBindUnsafeFilterColumn(t *testing.T) { t.Parallel() eq := "{{ jwt.org }}" diff --git a/internal/settings/settings.go b/internal/settings/settings.go index 596318bc..1ebf3089 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -92,11 +92,11 @@ type TenantConfig struct { // file that cannot be read or parsed is the one exception: boot refuses, // and a reload keeps the previous connection. type ClickHouseConfig struct { - // Addr is the native-protocol host:port (schema discovery, structured - // queries, pipes, /readyz). + // Addr is the native-protocol host:port (schema discovery, /readyz); + // queries and inserts go over the HTTP interface. Addr *string `json:"addr"` - // HTTPPort and HTTPScheme address the HTTP interface (ingest INSERTs and - // the raw-SQL proxy) on the same host as Addr. + // HTTPPort and HTTPScheme address the HTTP interface (structured queries, + // pipes, ingest INSERTs and the raw-SQL proxy) on the same host as Addr. HTTPPort *int `json:"http_port"` HTTPScheme *string `json:"http_scheme"` Database *string `json:"database"` @@ -109,12 +109,14 @@ type ClickHouseConfig struct { // material applies to whichever hop uses TLS. Paths, checked for shape // only — the files are read when the connection is (re)built. TLS *ClickHouseTLS `json:"tls"` - // Headers are set on every HTTP-interface request (ingest INSERTs, the - // raw-SQL proxy) ahead of WaveHouse's own credential and content-type - // headers, which therefore win. The native protocol carries none. + // Headers are set on every HTTP-interface request (structured queries, + // pipes, ingest INSERTs, the raw-SQL proxy) ahead of WaveHouse's own + // credential and content-type headers, which therefore win. The native + // protocol (discovery, /readyz) carries none. Headers map[string]string `json:"headers"` // MaxOpenConns and MaxIdleConns size the native driver's pool: each - // >= 1, open >= idle. + // >= 1, open >= idle. MaxOpenConns also caps the HTTP connections + // structured queries and pipes hold. MaxOpenConns *int `json:"max_open_conns"` MaxIdleConns *int `json:"max_idle_conns"` } diff --git a/internal/settings/validate.go b/internal/settings/validate.go index b7cf8a52..e0b24a85 100644 --- a/internal/settings/validate.go +++ b/internal/settings/validate.go @@ -528,13 +528,13 @@ func (v *validator) checkClickHouse(ch *ClickHouseConfig) { } v.checkClickHousePool(ch) // Both hops reach the same host, and each carries the credentials in the - // clear without TLS — the HTTP one on every insert and raw-SQL query, the - // native one in its handshake — so encrypting one alone is almost - // certainly not what the operator meant. One warning per direction. + // clear without TLS — the HTTP one on every query and insert, the native + // one in its handshake — so encrypting one alone is almost certainly not + // what the operator meant. One warning per direction. if ch.TLS != nil && ch.TLS.Enabled != nil && ch.HTTPScheme != nil { switch { case *ch.TLS.Enabled && *ch.HTTPScheme == "http": - v.warnf(FileConfig, "clickhouse.http_scheme", "clickhouse.tls.enabled is on but this HTTP hop is plaintext, and it carries the ClickHouse credentials on every insert and raw-SQL query") + v.warnf(FileConfig, "clickhouse.http_scheme", "clickhouse.tls.enabled is on but this HTTP hop is plaintext, and it carries the ClickHouse credentials on every query and insert") case !*ch.TLS.Enabled && *ch.HTTPScheme == "https": v.warnf(FileConfig, "clickhouse.tls.enabled", "clickhouse.http_scheme is https but the native hop is plaintext, and its handshake carries the ClickHouse password") } diff --git a/internal/stream/hub_test.go b/internal/stream/hub_test.go index 7eadf22a..27038cf9 100644 --- a/internal/stream/hub_test.go +++ b/internal/stream/hub_test.go @@ -214,8 +214,9 @@ func TestPairRow_Verdict(t *testing.T) { func rawEventCols(tb testing.TB, table, ts string, cols []string, data map[string]any) []byte { tb.Helper() // The positional line the ingest path publishes: one cell per column, in - // order, a column the record omits as null. Built here rather than through a - // production encoder because the wire shape is what these tests pin. + // order. A column data omits is written as null here, where production + // carries the value ClickHouse filled in; the hub forwards cells as they + // are, and the wire shape is what these tests pin. var buf bytes.Buffer buf.WriteByte('[') for i, c := range cols { @@ -1110,12 +1111,12 @@ func TestHub_RowFilter_BigIntegerExact(t *testing.T) { "the wire frame carries the exact digits, not a float64 rounding") } -// TestHub_RowFilter_TimestampInstantMatch: the wire carries ClickHouse's own -// rendering of a DateTime, and policy authors write the same zone-less spelling -// the query path wants. The filter compares them as instants because the row is -// parsed into the column's real storage before the predicate runs, so a -// different spelling of the same instant still matches; an operand the parser -// can't read withholds the row rather than guessing at it. +// TestHub_RowFilter_TimestampInstantMatch: policy authors write the zone-less +// spelling the query path wants, while the wire carries ClickHouse's RFC 3339 +// rendering. The filter compares them as instants because the row is parsed +// into the column's real storage before the predicate runs, so any spelling of +// the same instant matches; an operand the parser can't read withholds the row +// rather than guessing at it. func TestHub_RowFilter_TimestampInstantMatch(t *testing.T) { t.Parallel() p := &policy.Policy{ @@ -1136,7 +1137,7 @@ func TestHub_RowFilter_TimestampInstantMatch(t *testing.T) { hub.Broadcast(topic, rawEvent(t, "clicks", "t1", map[string]any{"created_at": "2026-06-21 04:00:00", "page": "/a"})) f, cols, _ := recvEvent(t, sub) - assert.NotEmpty(t, f.Data, "the wire rendering matches the zone-less constant") + assert.NotEmpty(t, f.Data, "a row in the constant's own spelling matches") // A different spelling of the same instant matches too: the comparison is // between parsed instants, not between bytes. diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go index c05f14bf..c3351bb8 100644 --- a/internal/typelayer/ingest.go +++ b/internal/typelayer/ingest.go @@ -316,11 +316,12 @@ func declineAll(n int, msg string) Batch { // countRecords recovers the input record count when chtypes returned no // per-row detail, so the caller still gets an index-aligned answer. JSONEachRow -// records are newline-separated and json.Marshal escapes any newline inside a -// value, so counting lines is exact for the bodies this package is handed. A -// CSV field may legally contain a raw newline, and a WithNames header is a -// line but not a record, so for those formats the fallback can OVER-count, -// which produces extra declined verdicts — never an extra acceptance. +// records are newline-separated and a raw newline inside a JSON string is +// illegal, so counting lines is exact for compact NDJSON and a re-framed array. +// A blank line, a pretty-printed object's inner lines, a CSV field holding a +// raw newline and a WithNames header each add a line that is no record, so the +// fallback can OVER-count, which produces extra declined verdicts — never an +// extra acceptance. func countRecords(body []byte, known int) int { if known > 0 { return known diff --git a/tests/e2e/sdk/ingest.test.ts b/tests/e2e/sdk/ingest.test.ts index 9eb0a0a6..af89a5f3 100644 --- a/tests/e2e/sdk/ingest.test.ts +++ b/tests/e2e/sdk/ingest.test.ts @@ -422,8 +422,9 @@ describe("Ingest", () => { // CONTRACT CHANGE: a column the caller's role may not write is no longer a // gateway 403 `column "x" not allowed for insert`. Column policy is enforced - // by compiling the role's own schema WITHOUT the denied columns, so naming one - // is ClickHouse's own per-record UNKNOWN_FIELD — a 400 with + // by compiling the role's own schema, where a denied column is re-declared + // MATERIALIZED of its default, so naming one is ClickHouse's own per-record + // UNKNOWN_FIELD — a 400 with // `exception_code: 117`, whose message does not say whether the column exists. it("refuses a denied insert column with ClickHouse's code 117", async () => { const currentPolicy = readPolicyFile(); @@ -491,7 +492,7 @@ describe("Ingest", () => { // CSV and TSV are new accepted formats. They are positional // in the table's declaration order — every wire column, in that order, with - // an empty field meaning "take the DEFAULT". An end-to-end assertion is the + // an empty CSV field meaning "take the DEFAULT". An end-to-end assertion is the // only one that catches a column-order bug: a mis-ordered body still answers // 200. it("ingests a complete positional CSV row end to end", async () => { @@ -546,9 +547,10 @@ describe("Ingest", () => { it("ingests a complete positional TSV row, and a short row is code 27", async () => { const id = testId(); // TSV spells "take the DEFAULT" as ClickHouse's own \\N (null, which - // input_format_null_as_default turns into the column default). An EMPTY TSV - // field is the empty string, which a DateTime64 cannot read — unlike CSV, - // where an empty field IS the default. + // input_format_null_as_default turns into the default of a non-Nullable + // column like received_timestamp; a Nullable one would store null). An + // EMPTY TSV field is the empty string, which a DateTime64 cannot read — + // unlike CSV, where an empty field IS the default. const res = await fetch(`${WH_URL}/v1/ingest?table=${T.clicks}`, { method: "POST", headers: { diff --git a/tests/integration/discovery_metadata_test.go b/tests/integration/discovery_metadata_test.go index b88df59d..97d3de7b 100644 --- a/tests/integration/discovery_metadata_test.go +++ b/tests/integration/discovery_metadata_test.go @@ -53,7 +53,8 @@ func TestDiscovery_MetadataAgainstRealClickHouse(t *testing.T) { assert.EqualValues(t, 4, byName["day"].Position, "a MATERIALIZED column still occupies a position") // Two claims this layer makes that no unit fake can reach: testutil hardcodes // kind="DEFAULT" whenever HasDefault, and no unit case sets MATERIALIZED. - // HasDefault is load-bearing — validation.go uses it to decide "not required". + // HasDefault is load-bearing — the SDK codegen reads has_default to make a + // field optional. assert.True(t, byName["day"].HasDefault, "MATERIALIZED is a non-empty default_kind") assert.NotEmpty(t, byName["day"].DefaultExpression, "and carries its expression") From 9da80ecdb11514e5326b5704b9482de961a29c18 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 09:20:42 -0400 Subject: [PATCH 46/70] docs: fix round-3 findings (Nullable nulls, dedupe example, stale mechanisms) - An explicit null, or TSV \N, takes the DEFAULT only on a non-Nullable column; on a Nullable one it stores NULL, which fails an insert check. An omitted column falls back to the type's default, NULL on a Nullable column, not an "implicit zero". - A raw SSE consumer sees an omitted column's slot holding its evaluated DEFAULT, not null. - The Enable Dedup example adds the event_id column the Quick Start table lacks; dedupe keys on a column. The Quick Start fetches the chtypes artifact before make dev. - AGENTS.md and architecture.md: chsql's shared helpers, the chconn getters, the envelope's wire columns, the test schema registry; the query-error docs no longer mention a native-driver exception. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 12 ++++++------ CHANGELOG.md | 6 +++--- docs/src/content/docs/access-control.mdx | 2 +- docs/src/content/docs/api.md | 15 ++++++++------- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/deployment.md | 2 +- docs/src/content/docs/development.md | 12 ++++++++++-- docs/src/content/docs/getting-started.md | 2 +- docs/src/content/docs/ingest-pipeline.md | 2 +- docs/src/content/docs/sdk/streaming.md | 2 +- docs/src/content/docs/settings-directory.mdx | 2 +- 11 files changed, 34 insertions(+), 25 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 64dc5f81..b3d500e8 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -32,13 +32,13 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive` for the one setting folded over every tenant served, `gapWindows` handing the sweeper each tenant's own gap window (a rejected tenant's as its folder last had it, unbounded for one rejected since boot) and the `mq.max_bytes_gb` reconcile each served tenant's byte budget, and `defaultPolicy` for the one setting that still follows tenant `0`, a flat directory's ops-gate admin role; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, the same hook's `Hub.Prune` ends the open streams of a tenant no longer served, and `wireCache`'s hook drops, through `LocalCache.Prune`, the cache version index of a tenant no longer served ([#262](https://github.com/Wave-RF/WaveHouse/issues/262))), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `New` wires only what the process's `roles` need (discovery, dedupe, auth verifiers, the hub bridge and keepalive per API process; the ingest worker per ingest process; the sweeper under its lease through `elected`, on the embedded MQ only); a process without `api` serves `api.NewOpsRouter` — probes, `/version`, metrics, and the settings reload behind the operator key alone. `cmd/wavehouse` and `tests/integration` both boot through it - **`auth/`** — JWT auth middleware: HMAC **or** JWKS verification with `alg` pinned to the active verifier, role extraction from a configurable claim path; always runs, never rejects (bad token → empty role + stashed reason). One verifier per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9): `Authenticator` keys them by `tenant.ID` — the request store's `settings.Store.Tenant()`, through an injected `TenantSource`; `tenant.Default` on the tenant-exempt routes — built from each tenant's `auth` block by `Reconfigure`, dropped by `Prune` once the tenant stops being served (rejected or removed), released by `Close`; the secrets (`Config`) are boot-level and shared. A JWKS key set is fetched off the boot and reload paths: until one has been stored the verifier is pending and a token-bearing request gets `503` + `Retry-After` from `api.refuseUnverifiable` (`auth.ErrVerifierPending`), never a `default_role` evaluation; refresh is library-managed (Eric, 2026-09-22), response capped at 1 MiB; the operator key's admin role is the request tenant's - **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index), and `RedisCache`, the Redis-compatible shared backend (random version tokens under the tenant's hash tag, one-round-trip lookups, bypass on failure behind a circuit breaker, deferred invalidations retried; selected by `cache.backend: redis`, configured by the boot config's `cache.redis` block — [#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every key carries the tenant (in `RedisCache`, after the key prefix: `:{}:…` for a version token, `:q::…` for a value); in `LocalCache` and the version index it leads ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for the caller's query key and its singleflight, escaped whole as the lead field of the stored key `|.||…`, where each raw table and scope name is escaped by `keyenc` (a `Namespace` carries them raw, so no caller escapes); the index holds a version per tenant, per (tenant, table) and per (tenant, table, scope), keyed by raw name and bumped in place (one entry per live namespace however often it is bumped, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)) — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` drops the tenant's index so its next key gets a process-unique generation, orphaning its every cached result in one step, pipe results included (no insert reaches a pipe result until [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); `Lookup` returns a `Snapshot` of the versions it read, taken before the handler chooses any input a bump invalidates — the tenant's connection included — and `Set` files the fill under it, so a write landing mid-query, or a reload moving the tenant to another address or database after the request took its connection, orphans the fill ([#382](https://github.com/Wave-RF/WaveHouse/issues/382)), and every backend runs the conformance suite `internal/testutil/cachetest`; the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the whole cache of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) -- **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (nil for a tenant on no pool, a `503`), `Target` (the tenant's own HTTP wiring over its pool's TLS config), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config. `Classify` (`errclass.go`) says what a failed ClickHouse request means for the request — `Unavailable`, `Denied`, `Rejected` (any unlisted exception code: the server read it and refused it), or `Unknown` (no code, no recognizable transport failure) — over the driver's error types and the HTTP interface's `HTTPError`; the ingest worker and the query handlers (`api/ch_errors.go` `writeCHError`, [#403](https://github.com/Wave-RF/WaveHouse/issues/403), [#271](https://github.com/Wave-RF/WaveHouse/issues/271)) both use it -- **`chsql/`** — dependency-free ClickHouse SQL helpers shared by `query`/`policy` (avoids an import cycle): `QuoteIdent` (backtick-quote every identifier) + `BindUnsafe` (reject names with a literal `?`) +- **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (the native connection schema discovery reads, nil for a tenant on no pool), `Target` (the tenant's own HTTP wiring over its pool's TLS config — what queries, pipes and inserts use; the zero `Target` for a tenant on no pool, a `503`), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config for the ingest worker and the raw-SQL proxy. `Classify` (`errclass.go`) says what a failed ClickHouse request means for the request — `Unavailable`, `Denied`, `Rejected` (any unlisted exception code: the server read it and refused it), or `Unknown` (no code, no recognizable transport failure) — over the driver's error types and the HTTP interface's `HTTPError`; the ingest worker and the query handlers (`api/ch_errors.go` `writeCHError`, [#403](https://github.com/Wave-RF/WaveHouse/issues/403), [#271](https://github.com/Wave-RF/WaveHouse/issues/271)) both use it +- **`chsql/`** — dependency-free ClickHouse SQL helpers shared by `query`, `policy`, `typelayer` and the ingest worker (avoids an import cycle): `QuoteIdent` (backtick-quote every identifier), `BindUnsafe` (reject names with a literal `?`), `EscapeStringParam` (the one `{p:String}` value encoding both read surfaces use) and `IntegerType`/`StrictInt`/`IntParam` (the strict round-trip cast an integer column's claims are compared through, in both renderers) - **`config/`** — YAML + env var config loading (cleanenv); strict on both sides (undeclared YAML key, unbound `WH_*` variable) and probes `data_dir` writability when a selected backend keeps state there (`NeedsDataDir`); `backends.go` holds each layer's `.backend` (the in-process value by default; `mq.backend` also takes `nats`, with its `mq.nats` sub-block of file-path-only credentials; `coord.backend` takes `nats`, whose `coord.nats` block names only the lease bucket and rides `mq.nats`'s connection (both blocks, their rules and warnings are `mq_nats.go`); `cache.backend` takes `redis`, whose sub-block is `cache_redis.go`; and `dedupe.backend` takes `dynamodb`, with its `dedupe.dynamodb` sub-block) and `Warnings`, the valid combinations boot logs at `WARN`; `config.go` holds `roles` (`Has(Role)`) and `instance_id`, and `Validate` refuses a role split the backends cannot serve (any split over the embedded MQ; `api` without `ingest`, or the reverse, over a local cache; `coord.backend=nats` without `mq.backend=nats`; `mq.backend=nats` with `coord.backend=local` in a process running `ingest`; a process running only `sweeper` under `mq.backend=nats`) — boot is the validator, there is no dry run - **`coord/`** — leases for work that must run in one process at a time (`Observer.Held` reads whether one is held without campaigning): `Coordinator.TryAcquire(ctx, name)` → a `Term` (fencing `Token`, strictly increasing per name; `Done`/`Err`, `ErrLost` on loss; `Resign`), `ErrHeld` while another holder's — or this coordinator's own — term is live; `RunElected` runs a loop only while holding its lease, resigning when the loop returns and campaigning again every `RetryPeriod`. `Local` is the in-process implementation (first taker wins, never expires; `Peer` is a second handle over the same table for tests); every implementation runs `coordtest.Conformance`. Imports only the standard library, so a distributed backend lives beside its connection: `coord.backend: nats` is `internal/mq/lease.go` (`ExternalNATS.Leases`), a key per lease in the operator's KV bucket, the KV revision as the fencing token, and expiry judged on the candidate's own clock (the same revision seen unchanged for 15s), never by a server TTL. `internal/app`'s `wireCoord` opens the one `coord.backend` selects and the sweeper runs through `RunElected` under the `sweeper` lease - **`dedupe/`** — `Deduplicator` interface (two-phase `Reserve`/`Commit`/`Release` over `Key{Table, ID}`; every backend passes the `dedupetest` conformance suite) → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant and table, pending claims in memory, committed ids stored with their expiry and deleted by an hourly background sweep along with the version-0 keys from before the table joined the key, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges) or `Dynamo` (one shared DynamoDB table, conditional `PutItem` claims; conformance-tested against dynamodb-local, selected by `dedupe.backend: dynamodb`; boot checks the table and never creates it outside dynamodb-local), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` or, gated on the table check (`Factory.Gated`), `Dynamo.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) - **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`, and starts a tenant over on a fresh registry when a reload moves it to another address or database: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version and default timezone, and fires an `OnRefresh` hook after every refresh it publishes and before the registry reports itself loaded (overlapping refreshes publish in the order they started: one that finishes after a later-started one has published is dropped, so an older snapshot never replaces a newer one), so a loaded tenant is a bound one — `typelayer.Engine.Bind` is its only consumer (Key Design Decision #21) -- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else; `row` is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced for that stored record (via `typelayer`'s `Table.IngestWith`), `columns` names its positions (the table's insertable columns, or the narrower list a column-restricted role produces); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs with `typelayer.InsertSettings()` plus `async_insert=0` (it never loads the artifact, so an ingest-only process needs none). In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) +- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else; `row` is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced for that stored record (via `typelayer`'s `Table.IngestWith`), `columns` names its positions (the table's wire columns, `typelayer.Table.WireColumns`: no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` column — or the narrower list a column-restricted role produces); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs with `typelayer.InsertSettings()` plus `async_insert=0` (it never loads the artifact, so an ingest-only process needs none). In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) - **`keyenc/`** — the one escaping composite keys are built from: `Escape` keeps `[A-Za-z0-9_-]` (exactly the tenant-id grammar, so a tenant id is its own escaped form) and writes every other byte as `%XX`, `Unescape` is `url.PathUnescape` (lenient: either hex case, and a byte left unescaped reads as itself, so a `%2D` an earlier build wrote still reads), `Join`/`AppendJoin` escape each field and put a separator between them (they panic on no fields, and on a separator the escaping could write or one outside ASCII) and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to and no field reaches a key unescaped: NATS subjects (`Join`/`Split` after the verbatim tenant) and the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — and the dedupe keys (`/
/`) use it; changing what it keeps orphans every stored key (an orphaned dedupe key lets a seen id through again), and on the shared backend, whose keys every process builds for itself, splits them between builds for the length of a rolling upgrade (a bump one build makes misses the entries the other filed, served until their TTL) - **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20), and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal, `ErrUnavailable` a broker that cannot be reached — both a `503`, with `Retry-After` `30` and `5`; `WithIdempotencyKey` makes a republish inside the queue's duplicate window a no-op), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). A broker whose ingest queue is split into units one consumer at a time owns implements `Sharded` too (`IngestUnits`, `ResetOrphaned`, `Unowned`; `ConsumerConfig.Units` narrows a consumer to some of them, and its consumer is a `Releaser` and a `Halter`, capped per unit by `ConsumerConfig.MaxHeld`, with `ErrConsumerMismatch` when an operator durable no longer fits), and `Message.OnSettled` runs a hook once, at the first ack or nak attempt, confirmed or not. Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, stream names, sequences, and ack floors are private to the implementations, whose subject tokens are escaped by the shared `internal/keyenc`: `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`, `deadletter.go`), which `internal/app` constructs and hands everything else as a `mq.Broker`, and `ExternalNATS` (`external.go`, `subject_nats.go`, `nats_topology.go`: an operator-owned cluster whose streams, durables and lease bucket it never creates, changes, purges or deletes; `lease.go` holds `coord.backend: nats`'s leases in that bucket), which `internal/app` constructs from the `mq.nats` block when `mq.backend` is `nats` ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every implementation passes the conformance suite in `internal/mq/mqtest` (`mqtest.Run`), which states the `Broker` contract as behavior; a new backend runs it from its own test, with `mqtest.Caps` only where its semantics legitimately differ - **`observability/`** — OpenTelemetry pipeline: `InitProvider` wires trace/metric/log providers via OTLP gRPC (each signal independently gated). A top-level `Prometheus` config block drives an optional `/metrics` scrape endpoint that runs independently of OTLP push — standalone (Alloy/Mimir scrape, no collector), alongside OTLP, or off. `NewLogger` produces a slog handler that fans out to stdout AND OTLP (stdout always 100%, OTLP sample-rate-aware). `TraceHandler` injects trace_id/span_id from active spans. `tracer.go` provides W3C trace context propagation over message headers (`InjectHeaders`/`ExtractHeaders` on a plain header map; `internal/mq` injects on every publish and extracts on the `Subscribe` path, so this package never sees a NATS type). @@ -130,7 +130,7 @@ Tooling notes (the non-obvious bits `make help` won't tell you): - **Table-driven tests**: Use `tests := []struct{ name string; ... }` with `t.Run(tt.name, ...)` for test cases. - **Shared mocks in `internal/testutil/`**: Use `MockPublisher` (records `Publish` and `DeadLetter`), `MockCache`, `MockDeduplicator`, `MockSubscriber`, `MockMessage`, `MockPurger`, `MockDeadLetterStats` instead of creating ad-hoc mocks. See `testutil/mocks.go`. - **JWT helpers**: Use `testutil.MakeJWT(t, claims)` and `testutil.MakeExpiredJWT(t, claims)` for auth tests. See `testutil/jwt.go`. -- **Schema helpers**: Use `testutil.NewTestSchemaRegistry(t, tables)` for schema-aware tests — it builds the registry through the real discovery path (`Refresh` against a mock ClickHouse connection), so timestamp specs are precomputed like production. +- **Schema helpers**: Use `testutil.NewTestSchemaRegistry(t, tables)` for schema-aware tests — it builds the registry through the real discovery path (`Refresh` against a mock ClickHouse connection), so the derived fields are computed exactly as in production. - **Cache backends**: every `cache.Cache` backend runs `cachetest.Run` (`internal/testutil/cachetest`), the backend-agnostic conformance suite; a behavior the contract promises goes there, not in one backend's tests. `RedisCache` runs it from `internal/cache/redis_integration_test.go` (`//go:build integration`, pinned Redis, Valkey, Dragonfly and Redis Cluster containers), which `make test-integration` includes. - **Write classifier cases**: `api.IsMutation`'s cases live in `internal/testutil/mutationtest`, shared by its unit test and `tests/integration/ismutation_test.go`, which checks each against ClickHouse's own parser; add a case there, not to either test. - **Policy helpers**: Use `policy.Static(p)` for a fixed `policy.Source` in tests. @@ -435,7 +435,7 @@ internal/app/ → Process wiring (build every component, run them unde internal/auth/ → JWT/JWKS authentication middleware (HMAC or JWKS, role extraction from claims) internal/cache/ → Query cache (interface, Ristretto L1, tenant-led version index, Redis-compatible shared backend) internal/chconn/ → ClickHouse pools, one per connection tuple among the served tenants (reconciled on settings reload) -internal/chsql/ → Shared ClickHouse SQL helpers (identifier quoting + bind-safety) +internal/chsql/ → Shared ClickHouse SQL helpers (identifier quoting, bind-safety, `{p:String}` value encoding, the strict integer-claim cast) internal/config/ → Configuration structs + loader internal/coord/ → Leases with fencing tokens (interface, in-process Local, RunElected, coordtest conformance suite) internal/dedupe/ → Optional deduplication (Reserve/Commit/Release interface; Pebble, DynamoDB) @@ -449,7 +449,7 @@ internal/policy/ → Access control policies (types, evaluation, Source) internal/query/ → Structured query AST + SQL builder internal/settings/ → Settings directory (validate, adopted snapshot + reload, watcher, embedded seed) internal/stream/ → SSE fan-out (event Hub: project once per role, Subscriber outbound queue, Bucket fan-out, keepalive Heartbeater wheel) -internal/typelayer/ → In-process ClickHouse parser (chtypes): ingest validation/coercion + row-level-security compilation +internal/typelayer/ → In-process ClickHouse parser (chtypes): ingest validation/coercion, insert checks + row-level-security compilation (typelayertest/ builds a test engine from the locked artifact) internal/tenant/ → Tenant id (type, grammar, reserved default, request header name) internal/testutil/ → Shared test helpers (mocks, JWT + schema helpers; logtest/ captures or silences the default logger; cachetest/ is the conformance suite every cache.Cache backend runs; mutationtest/ holds the shared write-classifier cases; storedir/ is the embedded broker's store directory in tests, removed once late consumer-state writes land) tests/ → Integration & E2E tests diff --git a/CHANGELOG.md b/CHANGELOG.md index 10b88f86..f0d6fe5e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -65,7 +65,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date`, `Date32` and `Time` are unaffected) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. -- **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, instead of walking a decoded record's keys. A column the role may not write stays in that schema as `MATERIALIZED` of its default, so naming it is refused while expressions that read it keep working and the stored row holds the table's default. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a checked column behaves exactly like omitting it. An `_eq` check also covers a column the role may not otherwise write: a record that omits it is filled with the required value and published, one that supplies exactly that value is accepted (**0.1.0 answered `403 column "x" not allowed for insert` to the correct value**), and any other value is `403 check failed for column "x"`. A role whose schema cannot be compiled this way, or that may write no column of the table, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After` (the cause is logged once a minute), rather than a `503` that a retry could not fix; a policy `check` on an `EPHEMERAL` column is still `403`. +- **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, instead of walking a decoded record's keys. A column the role may not write stays in that schema as `MATERIALIZED` of its default, so naming it is refused while expressions that read it keep working and the stored row holds the table's default. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a non-`Nullable` checked column behaves exactly like omitting it (on a `Nullable` one it stores `NULL`, which fails the check). An `_eq` check also covers a column the role may not otherwise write: a record that omits it is filled with the required value and published, one that supplies exactly that value is accepted (**0.1.0 answered `403 column "x" not allowed for insert` to the correct value**), and any other value is `403 check failed for column "x"`. A role whose schema cannot be compiled this way, or that may write no column of the table, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After` (the cause is logged once a minute), rather than a `503` that a retry could not fix; a policy `check` on an `EPHEMERAL` column is still `403`. - **WaveHouse now requires cgo, and supported platforms narrow to darwin/arm64, linux/amd64, linux/arm64** (BREAKING; `go.mod`, `scripts/build.sh`, `.goreleaser.yaml`, `deployments/Dockerfile`, `deployments/Dockerfile.goreleaser`, `Makefile`, `internal/config/config.go`, `config.yaml`, `chtypes.lock` (new), `scripts/fetch-chtypes.sh` (new), `.github/actions/setup-env/action.yml`, `.github/workflows/ci.yml`): the native type layer above needs cgo for `dlfcn` (no C library linked, no header). cgo is now unconditional: `CGO_ENABLED=0` is gone from every build path, and the `make audit-cgo` target that policed the old no-cgo build has been **removed** along with it (`make binary-analysis` is now `size` + `deadcode`). Because chtypes publishes artifacts only for darwin-arm64, linux-amd64 and linux-arm64, **Windows, FreeBSD and darwin/amd64 builds are discontinued** — `.goreleaser.yaml`'s matrix drops from 8 targets to 3, and the release archives/checksums/GHCR image narrow to match. The runtime image moves from an Alpine/musl builder + `distroless/static` to `golang:1.27-bookworm` (glibc, ships gcc) + `distroless/cc-debian12` (glibc + libstdc++, which the SDK's shared library needs) and bakes the pinned chtypes artifact into the image at `/opt/chtypes/artifacts` via a new `chtypes.lock` (exact file + sha256 per platform/line) and `scripts/fetch-chtypes.sh --frozen` wrapper, so the container has no first-request download. `go.mod` moves to `go 1.27`. Only processes running the `api` role load the artifact, so an ingest-worker-only or sweeper-only process needs none installed; the glibc requirement is 2.34 or later. New boot config: `clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY` lets an operator point at an explicit registry directory instead of the SDK's own search path (the shipped image instead sets the SDK's own `CHTYPES_REGISTRY` env var directly). CI's `unit`/`integration`/`e2e` jobs fetch and cache the pinned artifact (`setup-env`'s new `chtypes` input) and set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` so a missing artifact fails the job instead of silently skipping the chtypes-backed tests. `GOLANGCI_LINT_VERSION` bumped `v2.11.4` → `v2.13.2`: the `go 1.27` bump panics `v2.11.4`'s type checker on every package; `v2.13.0` is the oldest release whose changelog claims go1.27 support, but it panics in this tree for a different reason (`nilness`/`honnef.co/go/tools@v0.8.0-rc.1` crashing while analyzing a third-party dependency), fixed once that dependency moves past its release candidate in `v2.13.1`. The cross-toolchain approach this bullet originally described was replaced before landing — see the release-pipeline entry above. @@ -92,13 +92,13 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **NATS envelope v2: the row travels positionally, with the column names sent alongside** (BREAKING; `internal/ingest/{types,compact,worker}.go`, `internal/api/ingest.go`, `docs/src/content/docs/{api.md,architecture.md,ingest-pipeline.md}`): `EventMessage`'s `data` object is replaced by `format` (`"JSONCompactEachRow"`), `columns` (the table's declaration order) and `row` (one compact line — a positional JSON array). The `INSERT` the worker emits carries the column names once for a whole group instead of every row repeating every key (each NATS envelope still carries its own `columns`, since a message must stand alone), and a reader can tell a schema change mid-stream from a reordering. **In-flight NATS messages published by an older version are not readable by the new worker** — an envelope whose `format` is absent or unknown, or whose `columns` and `row` can't be paired (a length mismatch, an undecodable row), carries no way to say which value belongs to which column. Such an envelope is parked on the DLQ (see the Fixed entry below), never inserted. **Drain the ingest queue before deploying.** The worker groups a batch by column list, so a schema change mid-stream splits the INSERT rather than corrupting it, and writes `INSERT INTO {table} (cols) FORMAT JSONCompactEachRow` — the table still binds as a server-side `Identifier` parameter, while the column list, which has no such parameter, is quoted client-side by the same `chsql.QuoteIdent` the query builder uses. A positional row has one value per column and no way to say "absent", so a field the record omitted now rides as an explicit `null` in its slot; `input_format_null_as_default=1` (already the server default — set explicitly for one configured otherwise) turns that back into the column's default for a **non-nullable** column, matching what omitting the key did under `JSONEachRow`. **Transitional divergence, on `Nullable` columns only, and it is not what that setting controls:** ClickHouse stores an explicit `null` as `NULL` on a nullable column whatever the setting says — only an *absent* key ever took the default — so a `Nullable(T) DEFAULT …` column now stores `NULL` where it previously took its default. Verified against ClickHouse 26.6.3 (omitted key → default; explicit null → `NULL` at either setting). *Against a server explicitly running `input_format_null_as_default=0`*, the reverse also changes: an explicit `null` for a non-nullable column with a default now takes the default rather than failing the row into the DLQ, because WaveHouse pins the setting instead of inheriting it. On a default-configured server that was already the behavior. Every other column type behaves as before. The DLQ flow is unchanged — its payload is the new envelope. Row cells are copied as their original bytes rather than re-encoded at each hop, so a 64-bit id past 2^53 keeps every digit end to end. (Superseded for the omitted-field behaviour: the row is now ClickHouse's own export with `DEFAULT`s already evaluated, so a `Nullable(T) DEFAULT …` column takes its default again and the divergence above no longer exists — see the type-layer entry.) -- **SSE: the column list is announced as an `event: schema` frame, and rows arrive positionally** (BREAKING for raw consumers; `internal/stream/{hub,subscriber,metrics}.go`, `internal/api/stream.go`, `clients/ts/src/stream/sse.ts`, `docs/src/content/docs/{api.md,sdk/streaming.md}`): a data frame's `data` object is replaced by `row`, the compact array reduced to the caller's projected positions, and each connection is sent `event: schema` — `{"table_name", "columns"}` — before its first row and again whenever the projected list changes. The announcement is **per connection**: it goes out at subscribe time from the schema registry, so a client on a quiet table knows the shape before any row arrives, and a late joiner or a reconnect is told again. It deliberately carries **no** `id:` line — an empty one would clear the client's `Last-Event-ID` and cost the connection its resumption point, and a schema frame has no event position of its own to offer. Replay follows the identical contract with its own drift state, because it writes straight to the socket while live events queue behind it — sharing the connection's state would let a live announcement claim the slot and leave a replayed row ahead of it with nothing to zip against. **Known limitation, deferred to the schema-versioning work:** those two states are not reconciled when they disagree, so if a table's column set changes while a client is connected *and* that client gap-fill-replays across the change, live rows arriving after the replay may carry no fresh schema frame until the next drift or a reconnect ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The SDK's arity check catches most of it — a row whose **length** disagrees with the announced list is dropped rather than guessed at — but not a **same-length** change (a `RENAME COLUMN`, or a drop paired with an add), where values zip under the wrong names until the next announcement or a reconnect. So the residual case costs correctness, not only availability. The TypeScript SDK consumes the schema event and zips each row back into an object, so `.stream()`, `.liveQuery()` and `StreamEvent.data` are **unchanged** — two things are visible: a column the producer omitted now arrives as an explicit `null` rather than an absent key, and a row object has a **null prototype** (`Object.create(null)`), so a ClickHouse column legitimately named `__proto__` becomes an own property instead of vanishing into the inherited setter — at the cost of `row.hasOwnProperty(…)`, `` `${row}` `` and `row.constructor` no longer working on it. Use `Object.hasOwn(row, …)`. The client-side `.select(…)` projection now builds its row the same way: `projectColumns` used a plain object literal, so a `__proto__` column survived an unprojected stream and vanished from a projected one — the guarantee held for the transport but not for the path most callers use. **The SSE reader and writer changed together, so a version-skewed pair is a silently dead stream — upgrade both.** `@wavehouse/sdk` publishes independently of the server, so a pinned frontend against a self-scheduled backend is the normal shape, not an edge case. A **new SDK against an older server** never receives an `event: schema` frame, so `_columns` stays unset and every data frame is dropped — no `error` callback fires, the stream simply delivers nothing. The warnings are bounded (three per cause per connection, then a suppression line), so **a quiet console is not evidence the stream is healthy**. An **older SDK against a new server** reads the removed `data` key and yields `data: undefined` for every event. Neither surfaces as a catchable error. A **raw** SSE consumer (a hand-rolled `EventSource`) must keep the announced list and zip against it; the SDK drops — with a warning, never an error — a row it has no list for or whose length disagrees with one, rather than delivering values under guessed names. +- **SSE: the column list is announced as an `event: schema` frame, and rows arrive positionally** (BREAKING for raw consumers; `internal/stream/{hub,subscriber,metrics}.go`, `internal/api/stream.go`, `clients/ts/src/stream/sse.ts`, `docs/src/content/docs/{api.md,sdk/streaming.md}`): a data frame's `data` object is replaced by `row`, the compact array reduced to the caller's projected positions, and each connection is sent `event: schema` — `{"table_name", "columns"}` — before its first row and again whenever the projected list changes. The announcement is **per connection**: it goes out at subscribe time from the schema registry, so a client on a quiet table knows the shape before any row arrives, and a late joiner or a reconnect is told again. It deliberately carries **no** `id:` line — an empty one would clear the client's `Last-Event-ID` and cost the connection its resumption point, and a schema frame has no event position of its own to offer. Replay follows the identical contract with its own drift state, because it writes straight to the socket while live events queue behind it — sharing the connection's state would let a live announcement claim the slot and leave a replayed row ahead of it with nothing to zip against. **Known limitation, deferred to the schema-versioning work:** those two states are not reconciled when they disagree, so if a table's column set changes while a client is connected *and* that client gap-fill-replays across the change, live rows arriving after the replay may carry no fresh schema frame until the next drift or a reconnect ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The SDK's arity check catches most of it — a row whose **length** disagrees with the announced list is dropped rather than guessed at — but not a **same-length** change (a `RENAME COLUMN`, or a drop paired with an add), where values zip under the wrong names until the next announcement or a reconnect. So the residual case costs correctness, not only availability. The TypeScript SDK consumes the schema event and zips each row back into an object, so `.stream()`, `.liveQuery()` and `StreamEvent.data` are **unchanged** — two things are visible: a column the producer omitted now arrives as an explicit `null` rather than an absent key (superseded: it now arrives holding its evaluated `DEFAULT`, since the row is ClickHouse's own export — see the type-layer entry), and a row object has a **null prototype** (`Object.create(null)`), so a ClickHouse column legitimately named `__proto__` becomes an own property instead of vanishing into the inherited setter — at the cost of `row.hasOwnProperty(…)`, `` `${row}` `` and `row.constructor` no longer working on it. Use `Object.hasOwn(row, …)`. The client-side `.select(…)` projection now builds its row the same way: `projectColumns` used a plain object literal, so a `__proto__` column survived an unprojected stream and vanished from a projected one — the guarantee held for the transport but not for the path most callers use. **The SSE reader and writer changed together, so a version-skewed pair is a silently dead stream — upgrade both.** `@wavehouse/sdk` publishes independently of the server, so a pinned frontend against a self-scheduled backend is the normal shape, not an edge case. A **new SDK against an older server** never receives an `event: schema` frame, so `_columns` stays unset and every data frame is dropped — no `error` callback fires, the stream simply delivers nothing. The warnings are bounded (three per cause per connection, then a suppression line), so **a quiet console is not evidence the stream is healthy**. An **older SDK against a new server** reads the removed `data` key and yields `data: undefined` for every event. Neither surfaces as a catchable error. A **raw** SSE consumer (a hand-rolled `EventSource`) must keep the announced list and zip against it; the SDK drops — with a warning, never an error — a row it has no list for or whose length disagrees with one, rather than delivering values under guessed names. - **`cors.allowed_origins: []` now denies every browser origin instead of allowing all of them** (`internal/api/router.go`, `internal/settings/{validate,store}.go`, `docs/src/content/docs/settings-directory.mdx`; closes [#515](https://github.com/Wave-RF/WaveHouse/issues/515)): `corsMiddleware` treated an empty or nil allowlist as `["*"]`, so the natural spelling for "no origins" in a hand-edited, hot-reloadable `config.json` silently opened the API to every origin — [#508](https://github.com/Wave-RF/WaveHouse/pull/508) could only warn about it. An empty list is now an empty allowlist: no `Access-Control-Allow-Origin` (or any other CORS header) is sent to any origin, preflights get a bare `204`, and `Vary: Origin` is still emitted so a shared cache can't replay the headerless reject to an origin a later reload allows. A nil getter or nil slice (no settings source wired) denies the same way, so a missing source fails closed rather than open. `["*"]` is the only allow-all spelling. The `[]` validator warning is gone since the spelling now means what it says; the seed and the compose settings still ship `["*"]`, so nothing changes for a directory written by `wavehouse bootstrap`. - **The landing page's live demo now reads from the stats deployment's new WaveHouse Cloud backend** (`docs/src/components/LiveDemo.astro`, `docs/scripts/screenshot.mjs`): the GitHub-activity dogfood deployment behind the hero panel (Wave-RF/WaveHouse-Stats) moved off its self-managed AWS infrastructure onto WaveHouse Cloud, so `BASE_URL` — the origin `@wavehouse/sdk` queries in the visitor's browser — points at `https://iefrrvavd5akvphk7pq3.wavehouse.app` instead of `https://stats.wavehouse.dev`, ahead of the AWS stack being torn down. The `PUBLIC_WAVEHOUSE_STATS_URL` build-time override is unchanged, so a fork or staging docs build still redirects the panel without a code edit. **`DEMO_HOST` deliberately stays `stats.wavehouse.dev`** — the demo *site* is still served there and is still what the panel's chrome label and "Full demo" link should show; the migration splits the site from the API origin behind it, and the two constants now carry comments saying so. Verified against the new deployment before the switch: all five pipes the panel reads (`gh_summary`, `gh_activity_recent`, `gh_events_per_minute`, and the pre-#19 `gh_stars_total` / `gh_forks_total` fallbacks) return `200` with the same row shapes, the structured-query backfill fallback (`POST /v1/query?table=gh_events`) matches its old-backend response byte for byte, `GET /v1/stream?table=gh_events` opens an SSE stream, and CORS is unchanged (`Access-Control-Allow-Origin: *`, `X-Cache` exposed) so the cross-origin browser reads keep working from the docs site. The new backend is already the live ingest target — it reported more recent events than the old one at cutover (5,175 vs 5,038 over 7d) — which is the other half of why the panel had to follow it. `screenshot.mjs`'s `networkidle` note is retargeted to "the stats demo backend" rather than naming a host it no longer connects to. -- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. The reader's HTTP connections are capped per connection tuple (URL, user, password, database, TLS) at the largest `max_open_conns` among the tenants sharing it. The 64 MiB cap is new on these two paths, so a pipe (which has no row limit) returning more than 64 MiB now fails; `/v1/query` normally stays under it through its default row cap. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. +- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. The reader's HTTP connections are capped per connection tuple (URL, user, password, database, TLS) at the largest `max_open_conns` among the tenants sharing it, and its requests carry the tenant's `clickhouse.headers`; the settings warning about a plaintext HTTP hop (`internal/settings/validate.go`) now says it carries the credentials on every query and insert. The 64 MiB cap is new on these two paths, so a pipe (which has no row limit) returning more than 64 MiB now fails; `/v1/query` normally stays under it through its default row cap. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. - **WH001 (no hard-wrapped prose) now applies to every tracked Markdown file, with no carve-out** (`.github/.markdownlint.json` (deleted), `.claude/.markdownlint.json` (deleted), `.claude/skills/integration-astro-view-transitions/` (deleted), `.markdownlint-cli2.jsonc`, `.github/workflows/README.md`, `.claude/skills/pm-triage/references/routine.md`, `AGENTS.md`, `scripts/docs-prose.sh`, `.github/prompts/docs-review.md`, `docs/src/content/docs/claude-code.md`, `docs/src/content/docs/development.md`, `.claude/agents/docs-reviewer.md`): two path-scoped configs had switched WH001 off under `.github/` and `.claude/` ever since [#489](https://github.com/Wave-RF/WaveHouse/pull/489) introduced the rule — baked in from the start rather than added in response to a discovered problem — which left the repo documenting the rule three ways and disagreeing with itself: `CONTRIBUTING.md` promises contributors `make lint` enforces it *everywhere*, while `AGENTS.md` and the `.markdownlint-cli2.jsonc` header wrote up the carve-out. Not theoretical: on [#520](https://github.com/Wave-RF/WaveHouse/pull/520) a reviewer correctly flagged a hard-wrapped bullet in `.github/workflows/README.md`, an agent pointed at `"WH001": false` for that path and pushed back, and the reviewer recorded a *learning* never to flag WH001 there — the wrong invariant, learned off the wrong side of the contradiction ([#521](https://github.com/Wave-RF/WaveHouse/issues/521)). Both configs are deleted — each held nothing but the override, so the root `.markdownlint.json` governs again — and the 51 hard-wrapped paragraphs they were hiding are joined: 41 in `.github/workflows/README.md` and 10 in `.claude/skills/pm-triage/references/routine.md`, mechanical joins with no wording changed and every fenced block, table row, and heading byte-identical either side of the reflow. Deleted with them: the wizard-installed PostHog skill at `.claude/skills/integration-astro-view-transitions/` — 9 files, ~1,456 lines, including an 809-line `EXAMPLE.md` copied wholesale from `PostHog/context-mill`. Its integration job finished in [#277](https://github.com/Wave-RF/WaveHouse/pull/277), nothing in the repo calls it, and the docs-site setup it once described is documented where it belongs — in `docs/src/components/PostHog.astro` and this file. Keeping unowned third-party prose in the tree means content that drifts silently on every upstream bump and that nobody here reviews; it was also the single file that would have needed a special-case lint exclusion, so removing it is what lets WH001 apply with **no exception at all** rather than one documented one. Its two inventory rows in `claude-code.md` go with it, as does the now-dead `docs/posthog-setup-report.md` entry in the `scripts/docs-prose.sh` denylist (the wizard's other artifact, deleted back in [#502](https://github.com/Wave-RF/WaveHouse/pull/502)) and the copies of that denylist in `AGENTS.md` and `.github/prompts/docs-review.md`, which the script's header requires be kept in lockstep. Review of the change then turned up four more things the exclusion had been hiding, all fixed here: **WH001 has a blind spot** — `no-hard-wrapped-prose.mjs` classifies any line indented four or more spaces as an indented code block, so a *nested* list item is never joined, which left three hard-wrapped bullets in `.github/workflows/README.md` §"Adding a job" that the autofix could not see (unwrapped by hand; they were the last hard-wrapped prose paragraphs in the repo) and made `AGENTS.md`'s and `development.md`'s "a list item is joined as a unit" wrong for nested items (both now state the four-space caveat); the `scripts/docs-prose.sh` header told readers to keep its denylist in lockstep with **two** sibling copies when there are **three** — the missed one being `.claude/agents/docs-reviewer.md`, the gating subagent's own system prompt, which had in fact been silently out of sync for the whole life of the `posthog-setup-report.md` exclusion; the `.markdownlint-cli2.jsonc` header's "applies to every tracked Markdown file" was exact for WH001 but not WH002, which returns early on anything that isn't `.mdx`; and the job-graph diagram omitted `docs-deploy`'s `needs` edges from `unit`, `integration`, and `e2e`, contradicting invariant 2 three lines below it. The denylist also drops its `PERF-CLAIMS-REVIEW.md` entry — unlike the wizard artifact this one names a file that was **never tracked** at all, so it guarded a hypothetical; the list's other general cases are patterns (`*.draft.md`, `*.old.md`) that already cover a one-off review document, and a literal filename restated in four places is the outlier. `scripts/docs-prose.sh all` still resolves the same 27-file prose set. diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 1c1005ce..658fbf5f 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -286,7 +286,7 @@ Row filters apply on the structured-query path and, per subscriber, on the live Checks are compiled into **one chtypes filter** — `col = {p:String}` for `_eq`, `col IN (…)` for `_in`, AND-joined over every checked column, with an integer column's claim compared through the strict cast described under [JWT claim templating](#jwt-claim-templating) — and evaluated against the row ClickHouse's own parser produced, by the same engine that evaluates a row `filter`. So a check sees the **stored** value, after coercion and after `DEFAULT`s: what it admits is what the table will hold. - **If the request body includes the column**, its value must satisfy the check or the record is rejected with `403 check failed for column "x"` (`… for columns "x", "y"` when more than one is checked — the filter is AND-joined, so it names the set tested rather than inventing an attribution). Comparison is ClickHouse's, under the column's own type: on an integer, `Decimal`, `UUID` or `String` column a writer cannot forge a row for another tenant, because equal values store equal. A `Float32`/`Float64` column rounds the stored value and gives that exactness back — the row can land on a neighboring id. A check the engine cannot evaluate at all is a `422`, never a silent pass. -- **If the request body omits the column** — or sends an explicit `null`, which `input_format_null_as_default` resolves the same way — an `_eq` check **auto-injects** the claim-derived value, so clients can send just the business fields and let the policy stamp `user_id` and `tenant_id` from the token. It is implemented as a `DEFAULT` on the role's compiled schema, so a value the caller *does* send still wins. An `_eq` check also covers a column the role may not otherwise write (left out of `allow_columns`, or in `deny_columns`): a record that omits it is filled with the required value, one that supplies exactly that value is accepted, and any other value fails the check (`403`). An `_in` check has no single value to stamp, so the **table's own default** is what gets tested: the record is admitted if that default is in the claim-derived set and rejected if it is not. +- **If the request body omits the column** — or sends an explicit `null` on a non-`Nullable` column, which `input_format_null_as_default` resolves the same way — an `_eq` check **auto-injects** the claim-derived value, so clients can send just the business fields and let the policy stamp `user_id` and `tenant_id` from the token. On a `Nullable` column an explicit `null` is stored as `NULL`, which fails the check (`403`). It is implemented as a `DEFAULT` on the role's compiled schema, so a value the caller *does* send still wins. An `_eq` check also covers a column the role may not otherwise write (left out of `allow_columns`, or in `deny_columns`): a record that omits it is filled with the required value, one that supplies exactly that value is accepted, and any other value fails the check (`403`). An `_in` check has no single value to stamp, so the **table's own default** is what gets tested: the record is admitted if that default is in the claim-derived set and rejected if it is not. - **The check's column must be one a record can actually carry.** A `check` naming a column the table does not have, one ClickHouse computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is refused with a per-record `403` naming the column — on *every* insert by that role, until the policy or the table is corrected. None of the three can be enforced: the published row has one slot per wire column, so an injected value for a computed or unknown column is dropped on the way out, and an ephemeral column is never stored. Each would have answered `200` while enforcing nothing. `wavehouse validate` cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**. - **A claim literal the column cannot read fails closed, not loosely.** A `_eq` value that is not a legal literal for the column (`"1.0"` on a `UInt64`) cannot be compiled as that column's `DEFAULT`, so the injection is dropped and logged; the filter then judges the record as sent, which an absent column loses. On an integer column every record then fails the check — a `403`; on another column a supplied value is ClickHouse's code 53 `TYPE_MISMATCH` per row — a `422`. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index d4626de2..23200082 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -79,7 +79,7 @@ For SSE, streaming endpoints, or any handler that has already started writing th ### ClickHouse errors on the query paths -When ClickHouse fails a query on [`POST /v1/query`](#post-v1querytabletable--structured-query), [`/v1/pipes/{name}`](#getpost-v1pipesname--execute-named-pipe) or [`POST /v1/ops/query`](#post-v1opsquery--query-clickhouse), the status comes from **what kind of failure it was**, not from ClickHouse's HTTP status: ClickHouse answers a syntax error, a missing grant and an overloaded server alike with HTTP `500`. WaveHouse reads the ClickHouse exception code (the `X-ClickHouse-Exception-Code` header or the `Code: NNN.` in the message, or the native driver's exception) and answers with two extra fields alongside `error`: +When ClickHouse fails a query on [`POST /v1/query`](#post-v1querytabletable--structured-query), [`/v1/pipes/{name}`](#getpost-v1pipesname--execute-named-pipe) or [`POST /v1/ops/query`](#post-v1opsquery--query-clickhouse), the status comes from **what kind of failure it was**, not from ClickHouse's HTTP status: ClickHouse answers a syntax error, a missing grant and an overloaded server alike with HTTP `500`. WaveHouse reads the ClickHouse exception code (the `X-ClickHouse-Exception-Code` header or the `Code: NNN.` in the message) and answers with two extra fields alongside `error`: ```json {"error": "Code: 62. DB::Exception: Syntax error: …", "code": "clickhouse.rejected", "retryable": false} @@ -223,7 +223,7 @@ Every other route answers `404`, including every tenant route. Under `/v1/ops`, Validates a body of records against the ClickHouse schema for `{table}` and publishes each accepted one to the message queue. Returns immediately — ClickHouse insertion happens asynchronously via the batch consumer. A single-object body answers `{"ok":true}` (or `{"duplicate":true}` when dedup is on); every other body answers the [batch summary](#batch-ingest). -**The body goes to ClickHouse's own parser as-is.** WaveHouse never decodes a record: validation, type coercion, `DEFAULT` substitution and timestamp parsing are ClickHouse's own, running in-process via [chtypes](/deployment#chtypes-artifacts) (`internal/typelayer`) — the exact code path a real `INSERT` runs. A rejection therefore carries ClickHouse's own message and its numeric error number as `exception_code` rather than a WaveHouse-authored sentence, and there is no separate coercion table to keep in sync with the server. A rejected single record answers `{"error","exception_code"}` with no string `code`; only a refusal of the whole request (a `header=present` header naming a column the table lacks) carries both, `code: "clickhouse.rejected"` and `exception_code`. The SDK reports either as `HTTP_400`, with the number in `error.details`. +**The body goes to ClickHouse's own parser as-is.** WaveHouse never decodes a record: validation, type coercion, `DEFAULT` substitution and timestamp parsing are ClickHouse's own, running in-process via [chtypes](/deployment#chtypes-artifacts) (`internal/typelayer`) — the exact code path a real `INSERT` runs. A rejection therefore carries ClickHouse's own message and its numeric error number as `exception_code` rather than a WaveHouse-authored sentence, and there is no separate coercion table to keep in sync with the server. A rejected single record answers `{"error","exception_code"}` with no string `code`; only a refusal of the whole request (a `header=present` header naming a column the table lacks) carries both, `code: "clickhouse.rejected"` and `exception_code`. The SDK reports a rejected single record as `HTTP_400`, with the number in `error.details`. **`Content-Type` is required and authoritative**: it declares the format and the bytes never override it. @@ -255,7 +255,7 @@ The policy engine authorizes mutations by inspecting the columns being written. **What ClickHouse decides, and what WaveHouse decides.** Everything about a *value* is ClickHouse's: - A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED` and `ALIAS` columns — neither is ever part of a published row. (An `_eq` [insert check](/access-control#insert-checks) on a column the role may not write is the one exception: it accepts exactly the required value, and the row carries it.) An `EPHEMERAL` column is accepted as input only where the format names its columns (the JSON family and the `…WithNames` formats), the role may write it, a `DEFAULT` column reads it, and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column reads it; its value feeds that `DEFAULT` and is never stored, selected or published. Any other `EPHEMERAL` column is refused like an unknown one (code **117**): ClickHouse computes `MATERIALIZED` columns at insert time from the published row, which never carries the ephemeral value, so accepting it there would drop it silently. A positional CSV or TSV body (no header, or `header=absent`) carries the wire columns only. -- An omitted column, or an explicit `null` on one (WaveHouse pins `input_format_null_as_default`), takes its `DEFAULT` expression — evaluated by ClickHouse, including a volatile one like `now()` — or the type's implicit zero where none is declared, exactly as an `INSERT` naming fewer columns does. +- An omitted column takes its `DEFAULT` expression — evaluated by ClickHouse, including a volatile one like `now()` — or the type's default where none is declared (`NULL` on a `Nullable` column), exactly as an `INSERT` naming fewer columns does. An explicit `null` does the same on a non-`Nullable` column (WaveHouse pins `input_format_null_as_default`); on a `Nullable` column it stores `NULL`. - A coercion ClickHouse would make it makes here (a numeric string into an `Int*`, `"true"` into a `Bool`, an out-of-range integer wrapping); anything it would refuse fails synchronously in the ingest response with its real code, rather than surfacing later in the DLQ. `Nullable()` and `LowCardinality()` wrappers are transparent. WaveHouse decides only policy: whether the role may insert at all, and whether the record satisfies the role's [`check` clauses](/access-control#insert-checks) — evaluated by the same compiled-filter engine as row-level security, in the same parse that validates the record and against the row ClickHouse produced, so a check sees stored values rather than the payload's spelling. A record ClickHouse refuses reports that refusal, never a check result. A record chtypes cannot evaluate at all — as opposed to accepting or rejecting it — is **declined** (`422`), which is not a data verdict. @@ -326,13 +326,14 @@ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling Cl | Body | Outcome | | --- | --- | | every field, in order | accepted | -| an empty field (CSV) or `\N` (TSV) | that column takes its `DEFAULT` | +| an empty field (CSV) | that column takes its `DEFAULT` | +| `\N` (TSV) | `NULL` on a `Nullable` column; any other column takes its `DEFAULT` | | too few fields | rejected, code **27** — ClickHouse's own message, e.g. `Cannot parse input: expected ',' before: …` | | too many fields | rejected, code **117** — `Expected end of line` | | a header line, no `header` parameter | ClickHouse detects it and **consumes** it as a header: `total` and every `index` count data rows only | | a header line, `header=absent` | **not a header** — read as a data row, so it fails to parse (code 27) wherever a column cannot read its own name; the data rows after it still parse | -The messages are ClickHouse's own and differ between ClickHouse lines; branch on the `exception_code`. An empty **TSV** field is the empty string, not a default: `\N` is TSV's spelling for "take the default", and a `DateTime64` cannot read `""`. +The messages are ClickHouse's own and differ between ClickHouse lines; branch on the `exception_code`. An empty **TSV** field is the empty string, not a default: `\N` is TSV's `NULL`, which a non-`Nullable` column turns into its default, and a `DateTime64` cannot read `""`. A positional producer cannot self-describe, so a column-order change silently re-assigns values — pin it to the schema and re-check it after any `ALTER`, or send a header with `header=present`. @@ -669,7 +670,7 @@ id: 2026-03-24T12:00:01.456Z data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,"2026-03-24T12:00:01.456Z"]} ``` -A raw consumer must keep the most recent announced column list and zip each `row` against it; a value the record did not carry arrives as `null` in its slot rather than being omitted, so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you and still yields row objects — `.stream()` and `.liveQuery()` are unchanged. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. +A raw consumer must keep the most recent announced column list and zip each `row` against it; a column the record omitted still has its slot, holding its evaluated `DEFAULT` (or the type's default — `null` only on a `Nullable` column with none), so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you and still yields row objects — `.stream()` and `.liveQuery()` are unchanged. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. Each SSE connection is bound to a single `?table=`; to consume multiple tables, open one connection per table. @@ -901,7 +902,7 @@ The request that produced this envelope omitted `received_timestamp` (`DEFAULT n | `received_timestamp` | string | RFC 3339 nano timestamp when WaveHouse received the event. | | `format` | string | Row format. Always `JSONCompactEachRow` today; stated on the wire so a reader can tell an envelope it understands from one it doesn't. | | `columns` | string[] | The table's **wire** column names, in declaration order — what each position in `row` means (`internal/typelayer.Table.WireColumns`: the table's columns minus any `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column and minus any the role may not write — none of the three kinds is ever part of a published row). | -| `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order — the exact bytes ClickHouse's own writer produced for this stored row (`internal/typelayer`'s `Table.IngestWith`, via chtypes). A column the request body omitted carries its evaluated `DEFAULT` (or the type's implicit zero value where none is declared), not `null` — the same as a native `INSERT` naming fewer columns than the table has. `DateTime`/`DateTime64` values are ClickHouse's own rendering (see [Timestamp rendering](#timestamp-rendering)), and an out-of-range integer is wrapped the way a real `INSERT` wraps it. | +| `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order — the exact bytes ClickHouse's own writer produced for this stored row (`internal/typelayer`'s `Table.IngestWith`, via chtypes). A column the request body omitted carries its evaluated `DEFAULT`, or the type's default where none is declared (`null` only on a `Nullable` column) — the same as a native `INSERT` naming fewer columns than the table has. `DateTime`/`DateTime64` values are ClickHouse's own rendering (see [Timestamp rendering](#timestamp-rendering)), and an out-of-range integer is wrapped the way a real `INSERT` wraps it. | `columns` and `row` are only meaningful together: a reader that cannot pair them — a length mismatch, an undecodable row, a `columns` list naming one column twice — has no way to map a value to a column. Both readers also refuse an envelope whose `format` they do not recognize. Either way the SSE fan-out withholds such an envelope rather than guess, and the batch consumer parks it on the DLQ with `X-DLQ-*` headers — acking and dropping it only where the DLQ is switched off for that table, since it can never insert on retry. Both outcomes increment `wavehouse_ingest_poison_total`, separated by its `disposition` label (`parked` / `dropped`). diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index b5b20c22..a8c1e423 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -95,7 +95,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request ### `app/` — Process wiring - **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability (after which each `Config.Warnings` line is logged at `WARN`), ClickHouse pools, schema discovery, the dedupe stores, the MQ (embedded NATS with its ingest + DLQ streams, or the external NATS), cache, the lease coordinator, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. The boot config's `roles` decide which of them a process wires: every process gets the settings registry, observability, the MQ, the coordinator, the reload triggers and a listener; `api` adds schema discovery, the dedupe stores, streaming, auth and the full router; `ingest` adds the ingest worker; `sweeper` adds the sweeper on `mq.backend: embedded` only (under `nats` the streams' retention replaces it); the ClickHouse pools and the cache come with `api` or `ingest`. A process without `api` serves `api.NewOpsRouter` (probes, `/version`, the metrics path, and the settings reload behind the operator key alone, `wireOpsAuth`) on `server.port`. `config.Validate` refuses a role set the backends cannot serve (a split over the embedded MQ, ingest on a shared MQ over a local coordinator, a sweeper-only process on a shared MQ, or `api` without `ingest` and the reverse over a local cache), and `New` refuses a `Config` with no roles, which only one built without `config.Load` can have. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. A layer with a choice of implementation — `wireMQ`, `wireCache`, `wireDedupe`, `wireCoord` — picks it there and nowhere else, in a `switch` on the boot config's `.backend` with one case per backend; the default case refuses boot, which only a `config.Config` built without `config.Load` reaches, since `Validate` refuses a value no case handles. `wireCache` has two: `local`, the in-process `LocalCache`, and `redis`, the shared `RedisCache` built from the `cache.redis` block, which boots bypassed rather than failing when its server is unreachable. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where each would be redelivered every ack wait for as long as the tenant is away and would hold that tenant's ack floor, so the sweeper could purge none of its queue past it. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the whole cache — structured-query and pipe results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultPolicy` reads it through the registry, where a flat directory's tenant `0` is always served. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth, dedupe and cache hooks use too), ending the open streams of a tenant no longer served, and `wireCache`'s hook prunes the cache's version index the same way (`LocalCache.Prune`, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)), so a tenant no longer served stops holding it. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The ingest worker's queue, over a `mq.Sharded` broker (`mq.backend: nats`), is wrapped in `ingest.ClaimShards` over the same coordinator, so each process consumes only its share of the shards; the `nats` coordinator holds the ingest processes' membership leases there, since no sweeper is wired under `nats`. The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own, and runs only while its process holds the `sweeper` lease (`elected`, which wraps `coord.RunElected` over the coordinator `wireCoord` opens: `local` keeps leases in the process, so the one process always holds it; `nats` calls `ExternalNATS.Leases` on the MQ's own connection with the bucket `coordBucket` names — `coord.nats.bucket`, or `mq.DefaultNATSCoordBucket` of the subject prefix — and `instance_id` as the holder. `wireNATSMQ` hands the same bucket name to the topology, so boot waits for it with the streams. The coordinator is added after the MQ, so it closes first and resigns its terms while the connection is still up). The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe`'s `pebble` case builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) and then its table — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. `wireDedupe`'s `dynamodb` case is `wireDynamoDedupe`, in wire_dynamodb.go (below). `wireHTTP` hands the ingest handler `dedupe.lease` (`IngestHandler.DedupeLease`) whichever backend is chosen. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ`'s `nats` case builds an `mq.NATSConfig` from the boot config's `mq.nats` block and calls `mq.NewNATS`, which waits for the operator's topology under `New`'s context; it hands over no budget, since the operator's streams set every limit. Both cases end in `adoptMQ`, which registers the MQ's close and the system gauges. The `embedded` case hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. A layer with a choice of implementation — `wireMQ`, `wireCache`, `wireDedupe`, `wireCoord` — picks it there and nowhere else, in a `switch` on the boot config's `.backend` with one case per backend; the default case refuses boot, which only a `config.Config` built without `config.Load` reaches, since `Validate` refuses a value no case handles. `wireCache` has two: `local`, the in-process `LocalCache`, and `redis`, the shared `RedisCache` built from the `cache.redis` block, which boots bypassed rather than failing when its server is unreachable. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where each would be redelivered every ack wait for as long as the tenant is away and would hold that tenant's ack floor, so the sweeper could purge none of its queue past it. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`registryFor`, `chTargetFor`, `queryTimeout`, and `readConns`, the cap on the HTTP connections a tenant's pipes and structured queries hold), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is the zero `chconn.Target`, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the whole cache — structured-query and pipe results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultPolicy` reads it through the registry, where a flat directory's tenant `0` is always served. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth, dedupe and cache hooks use too), ending the open streams of a tenant no longer served, and `wireCache`'s hook prunes the cache's version index the same way (`LocalCache.Prune`, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)), so a tenant no longer served stops holding it. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The ingest worker's queue, over a `mq.Sharded` broker (`mq.backend: nats`), is wrapped in `ingest.ClaimShards` over the same coordinator, so each process consumes only its share of the shards; the `nats` coordinator holds the ingest processes' membership leases there, since no sweeper is wired under `nats`. The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own, and runs only while its process holds the `sweeper` lease (`elected`, which wraps `coord.RunElected` over the coordinator `wireCoord` opens: `local` keeps leases in the process, so the one process always holds it; `nats` calls `ExternalNATS.Leases` on the MQ's own connection with the bucket `coordBucket` names — `coord.nats.bucket`, or `mq.DefaultNATSCoordBucket` of the subject prefix — and `instance_id` as the holder. `wireNATSMQ` hands the same bucket name to the topology, so boot waits for it with the streams. The coordinator is added after the MQ, so it closes first and resigns its terms while the connection is still up). The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe`'s `pebble` case builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) and then its table — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. `wireDedupe`'s `dynamodb` case is `wireDynamoDedupe`, in wire_dynamodb.go (below). `wireHTTP` hands the ingest handler `dedupe.lease` (`IngestHandler.DedupeLease`) whichever backend is chosen. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ`'s `nats` case builds an `mq.NATSConfig` from the boot config's `mq.nats` block and calls `mq.NewNATS`, which waits for the operator's topology under `New`'s context; it hands over no budget, since the operator's streams set every limit. Both cases end in `adoptMQ`, which registers the MQ's close and the system gauges. The `embedded` case hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. - **wire_dynamodb.go** — `wireDedupe`'s `dynamodb` case, split out of wire.go so the e2e suite's coverage exclude for it (the e2e binary always runs Pebble dedupe, never DynamoDB) doesn't have to blanket wire.go itself: builds the same `dedupe.Stores` over `Dynamo.Tenant`, gated (`Factory.Gated`) on the table's check: boot runs `Dynamo.Check` (after `CreateTable`, when `dedupe.dynamodb.create_table` is on) whether or not any tenant has dedupe on. Boot is refused only for a misconfigured table (an error that is not `ErrUnavailable`) over a flat directory whose tenant has dedupe on; every other failure boots with the switched-on stores closed, the check retried until it passes by a background component that backs off from one second to thirty (a nested directory has no watcher, and a flat one's table can come good with no settings change). The `AfterAdopt` hook never runs the check, since it holds the lock that serializes reloads, and it does not wait on a tenant whose `dedupe.enabled` is unchanged either — `Managed.Apply`'s no-op fast path settles that case under its own read lock, so the hook only takes a store's write lock, and so waits for that tenant's in-flight `Reserve`/`Commit`/`Release` calls to finish, on a genuine flip. It applies every store against the last check's result, so a tenant a reload switches on fails closed meanwhile, and wakes the retry, so a reload still retries at once. It has no Pebble gauges. - **wire_nats.go** — the `nats` cases of `wireMQ` and `wireCoord` (`wireNATSMQ`, `wireNATSCoord`, and the lease bucket's name, `coordBucket`), split out of wire.go for the same reason as `wire_dynamodb.go`: the e2e binary never runs `mq.backend: nats`. `wireNATSMQ` hands the topology check the dedupe lease and, in a process running `api`, warns at boot and after every reload about a tenant with dedupe on whose finite retention is under the partitions' duplicate window (`ExternalNATS.DuplicateWindow`), and runs `ExternalNATS.CheckReplayWindows(servedGapWindows(...))` at boot and from an `AfterAdopt` hook, the check of a gap window longer than the history keeps that the sweeper made on the embedded MQ. diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index b18f648f..8a5196d1 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -645,7 +645,7 @@ For local development, `docker compose -f deployments/compose/dependencies.yaml WaveHouse uses a **Bring Your Own Schema** model. You create your tables in ClickHouse with whatever columns and engines you need. WaveHouse discovers the schemas automatically via `system.columns` and validates ingest data against them — see [Schema Validation](/api#post-v1ingesttabletable--ingest-data) for the rules a record must satisfy. -Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's implicit zero value where none is declared, evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). +Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's default where none is declared (`NULL` on a `Nullable` column), evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). Example table: diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index c52b728c..ef36edf0 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -69,6 +69,7 @@ This is the fastest way to get a fully functional local environment: git clone https://github.com/Wave-RF/WaveHouse.git cd WaveHouse make tools +scripts/fetch-chtypes.sh # once per machine (see above) # 2. Start ClickHouse (the only external dependency) docker compose -f deployments/compose/dependencies.yaml up -d clickhouse @@ -245,9 +246,16 @@ curl -s -X POST http://localhost:8080/v1/ops/query \ # set "enabled": true under "dedupe" in ./settings/config.json ``` -The key hot-reloads, so once the server is running you can toggle it by editing `config.json` — no restart. Records dedupe on their `event_id` field by default; the same file overrides the field globally or per table (see [Settings Directory — Deduplication](/settings-directory#deduplication)). +The key hot-reloads, so once the server is running you can toggle it by editing `config.json` — no restart. Records dedupe on their `event_id` column by default; the same file overrides the column globally or per table (see [Settings Directory — Deduplication](/settings-directory#deduplication)). -Then include the dedup field in your ingest body: +The Quick Start `clicks` table has no `event_id` column, and a record naming a column the table lacks is refused (code 117), so add one first; ingest sees it after the next schema refresh (60 seconds by default): + +```bash +docker compose -f deployments/compose/dependencies.yaml exec clickhouse \ + clickhouse-client --query "ALTER TABLE clicks ADD COLUMN IF NOT EXISTS event_id String" +``` + +Then include it in your ingest body: ```bash curl -s -X POST "http://localhost:8080/v1/ingest?table=clicks" \ diff --git a/docs/src/content/docs/getting-started.md b/docs/src/content/docs/getting-started.md index 24c62f55..99f028e6 100644 --- a/docs/src/content/docs/getting-started.md +++ b/docs/src/content/docs/getting-started.md @@ -119,4 +119,4 @@ The handful of things that most often trip up a first session — each is expect ## Going further - **Validate JWTs**: set `WH_AUTH_JWT_SECRET=` (the middleware always runs; without a secret every request is the policy `default_role`) and replace the shipped trial policy (`deployments/compose/settings/policies.json`) with a least-privilege one — see [API Reference — Authentication](/api#authentication) and [Access Control](/access-control). -- **Enable deduplication**: set `dedupe.enabled` to `true` in the settings directory's `config.json` (it hot-reloads, no restart) — records dedupe on their `event_id` field by default; pick a different field (globally or per table) in the same file — see [Settings Directory — Deduplication](/settings-directory#deduplication). +- **Enable deduplication**: set `dedupe.enabled` to `true` in the settings directory's `config.json` (it hot-reloads, no restart) — records dedupe on their `event_id` column by default (add one to the table); pick a different column (globally or per table) in the same file — see [Settings Directory — Deduplication](/settings-directory#deduplication). diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index beccb4a4..d6c0d3bf 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -59,7 +59,7 @@ flowchart LR Note the embedded broker's stream is **dual-use**: it is both the durable buffer feeding the worker and the replay buffer that SSE clients gap-fill from. That is why a custom sweeper exists there instead of plain work-queue auto-deletion; `mq.backend: nats` splits the two roles instead, with work-queue partitions and a separate history stream (see [Scaling out](#scaling-to-multiple-instances)). :::note[Omitted columns take their real DEFAULT, not `null`] -The batch that reaches `insertToClickHouse` is not assembled from the request body — it is the bytes `IngestWith` returned for each accepted record, produced by ClickHouse's own writer. An omitted field's `DEFAULT` (or the type's implicit zero) was evaluated before that line existed, so a `Nullable(T) DEFAULT …` column takes its default exactly as an `INSERT` naming fewer columns would. Verified on ClickHouse 26.8. +The batch that reaches `insertToClickHouse` is not assembled from the request body — it is the bytes `IngestWith` returned for each accepted record, produced by ClickHouse's own writer. An omitted column's `DEFAULT` (or the type's default — `NULL` on a `Nullable` column with none) was evaluated before that line existed, so a `Nullable(T) DEFAULT …` column takes its default exactly as an `INSERT` naming fewer columns would. Verified on ClickHouse 26.8. ::: :::note[Insert settings pinned] diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index 2d56ba43..e082225a 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -97,7 +97,7 @@ interface StreamEvent { } ``` -`data` is a row **object**, as it always has been — but the wire underneath is positional. The server sends the column list in its own `event: schema` frame — before the first row, and again whenever the list drifts on the **live** path (with one exception after a gap-fill, below) — and each row as a JSON array; the SDK keeps the announced list and zips every row against it, so this shape is unchanged and nothing in your code moves. It matters in two places. The row object has a **null prototype**: a ClickHouse column may legitimately be named `__proto__`, and on an ordinary object that assignment hits the inherited setter and the value disappears — so the SDK builds each row with `Object.create(null)`. Property access, spreading, `JSON.stringify` and destructuring all behave normally; what does not is anything inherited from `Object.prototype`, so use `Object.hasOwn(row, "x")` rather than `row.hasOwnProperty("x")`, and don't rely on `` `${row}` `` or `row.constructor`. (`liveQuery`'s REST backfill half still yields ordinary objects.) And a **raw** SSE consumer (a hand-rolled `EventSource`) must do the zipping itself — see [the wire format](/api#get-v1stream--server-sent-events-stream). A column the producer omitted is no longer `null` on the wire: WaveHouse's ingest validation runs ClickHouse's own parser in-process, which evaluates the column's `DEFAULT` (or its implicit zero value) before the row is published — the same as a native `INSERT` naming fewer columns than the table has. +`data` is a row **object**, as it always has been — but the wire underneath is positional. The server sends the column list in its own `event: schema` frame — before the first row, and again whenever the list drifts on the **live** path (with one exception after a gap-fill, below) — and each row as a JSON array; the SDK keeps the announced list and zips every row against it, so this shape is unchanged and nothing in your code moves. It matters in two places. The row object has a **null prototype**: a ClickHouse column may legitimately be named `__proto__`, and on an ordinary object that assignment hits the inherited setter and the value disappears — so the SDK builds each row with `Object.create(null)`. Property access, spreading, `JSON.stringify` and destructuring all behave normally; what does not is anything inherited from `Object.prototype`, so use `Object.hasOwn(row, "x")` rather than `row.hasOwnProperty("x")`, and don't rely on `` `${row}` `` or `row.constructor`. (`liveQuery`'s REST backfill half still yields ordinary objects.) And a **raw** SSE consumer (a hand-rolled `EventSource`) must do the zipping itself — see [the wire format](/api#get-v1stream--server-sent-events-stream). A column the producer omitted is no longer always `null` on the wire: WaveHouse's ingest validation runs ClickHouse's own parser in-process, which evaluates the column's `DEFAULT` (or its type's default — `null` only for a `Nullable` column with none) before the row is published — the same as a native `INSERT` naming fewer columns than the table has. Row values of top-level `DateTime`/`DateTime64` columns inside `data` (not timestamps nested in `Array`/`Map`/`Tuple` columns) arrive as RFC 3339 in UTC, whatever zone the column declares — `"2026-06-21T04:00:00.123Z"` — matching what `/v1/query` returns for the same row byte-for-byte, by construction (see [Timestamp rendering](/api#timestamp-rendering)). `new Date(value)` parses it directly, and because the stream's `timestamp` field and a timestamp column use the same form, the SDK's comparisons between them (the live-query dedupe below, client-side `.where()` on a timestamp column) are by instant, not by spelling. A `DateTime64(3)` on a whole second arrives as `…:00.000Z`, and an event published before an upgrade, or by an older instance during a rolling deploy, replays in the spelling it was published in. diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 6f418704..38d01224 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -197,7 +197,7 @@ Every per-tenant dedupe knob lives here. Where the seen ids are kept (`dedupe.ba The `clickhouse` block is the connection wiring, minus the password. A reload applies it to every consumer (schema discovery, structured queries, pipes, `/readyz`, the ingest worker's HTTP `INSERT`s, and the raw-SQL proxy — the HTTP-side ones re-read the target per request). A change to `addr`, `database`, `username` or the `tls` block moves the tenant to the pool of its new tuple, opened for it when no served tenant has that tuple; a change to `max_open_conns` or `max_idle_conns` resizes its pool; a change to `http_port`, `http_scheme`, `headers` or `query_timeout` needs no new connection. A replaced connection stays open for the longest `query_timeout` among the tenants that were on it, so in-flight queries finish. The change is **unconditional**: the adopted settings are the authority, so an address that isn't reachable is applied all the same and shows up where reachability already does — schema discovery retries and logs, `/readyz` fails, queries return errors — until the next reload fixes it. Two things are refused instead: a `tls` block whose certificate files cannot be loaded (unreadable, not PEM, or a `cert_file` and `key_file` that do not pair), and a pool that would put the process over the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) — a `max_open_conns` raised above it keeps the pool at its size, and a new connection tuple that would cross it is not opened. An option the ClickHouse driver refuses to open is refused the same way, though validation already excludes every one it knows of. A refused tenant keeps the pool and wiring it had until the next reload, or, newly served with no pool to keep, stays on none and its queries answer `503`. The reload itself still reports `adopted` — the refusal is an `ERROR` log line, not a finding — and the next reload retries it; at boot either refuses to start. Validation checks shape only (`host:port`, port range, scheme, non-empty database and user, timeout `>= 1`, the `tls` block's shape without opening its paths, header names and values, pool sizes); reachability is a runtime concern, so `wavehouse validate` needs no ClickHouse and no certificate files. At boot the address is dialed lazily, as before: an unreachable ClickHouse degrades `/livez` and retries rather than refusing to start. -**TLS.** There are two hops and two switches: `tls.enabled` puts the native-protocol connection (`addr`) on TLS, and `http_scheme: "https"` does the same for the HTTP interface (`http_port`). The rest of the `tls` block — the authority bundle, a client certificate, `insecure_skip_verify`, `server_name` — applies to whichever hop uses TLS, so a ClickHouse behind a private authority needs `ca_file` once for both. The certificate files are read when a pool opens: at boot, and on a reload that names a tuple no open pool has — a changed `tls` block, `addr`, `database` or `username`. A reload that changes only `http_port`, `http_scheme`, `headers`, `query_timeout` or the pool sizes reuses the material already loaded, so a file replaced in place is picked up the next time a pool for its tuple opens, or by a restart, like a rotated secret. Both switches move the hop to ClickHouse's TLS listeners, so the ports move too: `addr` to the secure native port (`9440` by default) and `http_port` to the HTTPS one (`8443`), as the example above does. Validation warns when only one hop is on TLS: a plaintext HTTP hop carries the credentials on every insert and raw-SQL query, and a plaintext native hop sends the password in its handshake. In the container images these are container paths: bind-mount the bundle (`-v /srv/clickhouse-ca.pem:/etc/wavehouse/clickhouse-ca.pem:ro`) and make it readable by UID 65532, like the settings directory. +**TLS.** There are two hops and two switches: `tls.enabled` puts the native-protocol connection (`addr`) on TLS, and `http_scheme: "https"` does the same for the HTTP interface (`http_port`). The rest of the `tls` block — the authority bundle, a client certificate, `insecure_skip_verify`, `server_name` — applies to whichever hop uses TLS, so a ClickHouse behind a private authority needs `ca_file` once for both. The certificate files are read when a pool opens: at boot, and on a reload that names a tuple no open pool has — a changed `tls` block, `addr`, `database` or `username`. A reload that changes only `http_port`, `http_scheme`, `headers`, `query_timeout` or the pool sizes reuses the material already loaded, so a file replaced in place is picked up the next time a pool for its tuple opens, or by a restart, like a rotated secret. Both switches move the hop to ClickHouse's TLS listeners, so the ports move too: `addr` to the secure native port (`9440` by default) and `http_port` to the HTTPS one (`8443`), as the example above does. Validation warns when only one hop is on TLS: a plaintext HTTP hop carries the credentials on every query and insert, and a plaintext native hop sends the password in its handshake. In the container images these are container paths: bind-mount the bundle (`-v /srv/clickhouse-ca.pem:/etc/wavehouse/clickhouse-ca.pem:ro`) and make it readable by UID 65532, like the settings directory. **Headers.** `clickhouse.headers` rides on every HTTP-interface request: routing or identification metadata for a proxy or gateway in front of ClickHouse. Values are stored in the file as written, so a credential placed there is only as protected as the settings directory itself; secrets stay in boot config. WaveHouse's own headers — `Content-Type` and the `X-ClickHouse-User` / `X-ClickHouse-Key` credentials — are set after them and win. Naming `X-ClickHouse-User` or `X-ClickHouse-Key` is a validation error, and so is `Authorization`: ClickHouse's HTTP interface reads it as Basic credentials, a second and conflicting credential path, so a gateway that authenticates that way needs a header name of its own. The native protocol carries no headers. From 03433540047327fdb9fbbf8eca20335170dc99ed Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 09:37:02 -0400 Subject: [PATCH 47/70] fix(ingest): blank layout newlines after a JSON array's closing bracket A newline after the closing `]` was left in place, so a whole-batch decline without per-record detail counted it as a phantom record. It is now blanked like every other layout newline, leaving exactly elements-1. Also say how the stream path binds a Predicate's values: as {pN:String} parameters, the same as the query builder. Co-Authored-By: Claude Opus 5.5 --- internal/api/ingest_framing.go | 17 +++++++++++------ internal/api/ingest_framing_test.go | 2 +- internal/policy/policy.go | 6 +++--- 3 files changed, 15 insertions(+), 10 deletions(-) diff --git a/internal/api/ingest_framing.go b/internal/api/ingest_framing.go index fb55a4b3..eb9b4442 100644 --- a/internal/api/ingest_framing.go +++ b/internal/api/ingest_framing.go @@ -35,11 +35,12 @@ import ( // closing bracket shares that record's line and the reader cannot resync // past it. JSONEachRow needs no brackets, so removing them costs nothing and // makes every position salvageable, first and last included; -// - every other newline outside a string becomes a space. Not cosmetic: when -// chtypes declines a whole batch without per-record detail, the type layer -// counts the body's lines to answer each record, so a pretty-printed array -// would come back with one phantom declined record per line of layout. -// This leaves exactly elements-1 newlines, so that count stays right. +// - every other newline outside a string, before or after the brackets too, +// becomes a space. Not cosmetic: when chtypes declines a whole batch +// without per-record detail, the type layer counts the body's lines to +// answer each record, so a pretty-printed array would come back with one +// phantom declined record per line of layout. This leaves exactly +// elements-1 newlines, so that count stays right. // // A raw newline inside a string is illegal JSON, so leaving those alone costs // nothing and keeps the caller's bytes the caller's. @@ -57,7 +58,11 @@ func reframeArray(b []byte) (elements int, err error) { for i := range b { c := b[i] if closed { - if c != ' ' && c != '\t' && c != '\n' && c != '\r' { + switch c { + case ' ', '\t': + case '\n', '\r': + b[i] = ' ' + default: return 0, errAfterArray } continue diff --git a/internal/api/ingest_framing_test.go b/internal/api/ingest_framing_test.go index 632f4ebe..01d827d4 100644 --- a/internal/api/ingest_framing_test.go +++ b/internal/api/ingest_framing_test.go @@ -89,7 +89,7 @@ func TestReframeArray(t *testing.T) { { name: "whitespace after the array is layout", body: "[{\"a\":1}] \r\n\t", - want: " {\"a\":1} \r\n\t", + want: " {\"a\":1} \t", count: 1, ok: true, }, {name: "a truncated array does not balance", body: `[{"a":1}`, err: errUnterminatedArray}, diff --git a/internal/policy/policy.go b/internal/policy/policy.go index f01fd82e..66836de7 100644 --- a/internal/policy/policy.go +++ b/internal/policy/policy.go @@ -356,9 +356,9 @@ func (rp *ResolvedPermissions) HasRowFilter() bool { // canonical form — see resolveTemplate/CanonicalScalar). // // It is exported because the stream path evaluates it outside this package -// (Predicates). The values are bound as typed parameters there, exactly as they -// are bound as query parameters here — neither surface ever splices one into -// expression text. +// (Predicates). The values are bound as {pN:String} parameters there, exactly +// as they are bound here — neither surface ever splices one into expression +// text. type Predicate struct { Column string Op string From 56ba58fd988242f36e3caa0e86095533d97d4228 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 10:20:29 -0400 Subject: [PATCH 48/70] chore: correct comments contradicted by measurement on 26.8 - RFC 3339 against a DateTime is read on 26.8.15.10 (its cast_string_to_date_time_mode defaults to best_effort) and refused only on 24.8 or under basic; bind.go said 26.8 refused it. - An unreadable constant on a Float column is code 72, not 53 (53 is the Decimal case); an integer column's strict cast answers false. - Per-record ingest codes: no code 6 (out-of-range integers wrap); a header line under header=absent fails with the reading column's code. - The raw-SQL proxy takes one statement per request (code 62), and ClickHouse's HTTP status is not a flat 500. - TSDoc: exception_code is absent on a 422 decline, and a single-object refusal carries no string code; sql() takes one statement. Co-Authored-By: Claude Opus 5.5 --- clients/ts/src/sql.ts | 5 +++-- clients/ts/src/types.ts | 14 +++++++++----- internal/api/content_type.go | 5 +++-- internal/api/ingest.go | 8 +++++--- internal/api/query.go | 15 +++++++-------- internal/auth/auth_test.go | 6 +++--- internal/policy/canonical.go | 5 +++-- internal/policy/policy_test.go | 4 ++-- internal/query/bind.go | 7 ++++--- internal/stream/hub_test.go | 6 +++--- internal/stream/roweval.go | 5 +++-- internal/typelayer/filter.go | 4 ++-- internal/typelayer/filter_test.go | 4 ++-- internal/typelayer/ingest.go | 2 +- 14 files changed, 50 insertions(+), 40 deletions(-) diff --git a/clients/ts/src/sql.ts b/clients/ts/src/sql.ts index 48a81501..f1d14727 100644 --- a/clients/ts/src/sql.ts +++ b/clients/ts/src/sql.ts @@ -13,8 +13,9 @@ import type { HttpContext, OpsRequestOptions, Result } from "./types.js"; * the structured query builder (`wh.from(table)...`) instead. * * The server proxies the SQL string verbatim to ClickHouse's HTTP interface, - * so any ClickHouse-accepted statement works — including multi-statement - * input (`SELECT 1; TRUNCATE t`) and arbitrary DDL/DML/SYSTEM verbs. + * so any single statement ClickHouse accepts works, arbitrary DDL/DML/SYSTEM + * verbs included. ClickHouse's HTTP interface refuses multi-statement input + * (`SELECT 1; TRUNCATE t` is code 62): send one statement per call. * * **JSON-row contract.** This helper returns `Result` and assumes the * response is the standard `FORMAT JSON` envelope (or an empty body for diff --git a/clients/ts/src/types.ts b/clients/ts/src/types.ts index 1b82c924..63989d5d 100644 --- a/clients/ts/src/types.ts +++ b/clients/ts/src/types.ts @@ -307,11 +307,15 @@ export interface InsertRecordResult { error?: string; /** * ClickHouse's own numeric error code, present only when the server's parser - * is what refused the record — 117 unknown field, 27 or 6 a value the column - * cannot read. Absent for a gateway rejection (a failed policy check, a missing - * dedupe id), so `exception_code !== undefined` means "ClickHouse answered". - * The same name carries it on a whole-request error body, beside the string - * `code` class (reachable as `error.details`). + * is what refused the record — 117 unknown field, 27 or 26 input it cannot + * parse, or a type-specific code such as 41 for a `DateTime` (an out-of-range + * integer is not refused: it wraps). Absent for a gateway rejection (a failed + * policy check, a missing dedupe id, a `422` decline), so + * `exception_code !== undefined` means "ClickHouse answered". + * A single-object insert's refusal carries the same name in its error body, + * reachable as `error.details.exception_code` (the SDK's `error.code` is then + * `HTTP_400`); so does a header-format body refused as a whole, which also + * carries the string `code` class `clickhouse.rejected`. * * 117 also covers **a column the caller's role may not write**. Column policy * is enforced by compiling the role's own schema, where a denied column is diff --git a/internal/api/content_type.go b/internal/api/content_type.go index 85c5a614..cb261a43 100644 --- a/internal/api/content_type.go +++ b/internal/api/content_type.go @@ -70,8 +70,9 @@ const ( // FormatTSVWithNames is FormatCSVWithNames' tab-separated twin. FormatTSVWithNames // FormatCSVPositional is `text/csv; header=absent`: strictly positional, - // detection off, so a header line is one record that fails to parse with - // ClickHouse's code 27. + // detection off, so a header line is one record that fails to parse, with + // the code of the column that cannot read its own name (27 for an integer, + // 72 for a float). FormatCSVPositional // FormatTSVPositional is FormatCSVPositional's tab-separated twin. FormatTSVPositional diff --git a/internal/api/ingest.go b/internal/api/ingest.go index 708799c8..b5943c41 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -140,7 +140,8 @@ type recordResult struct { Error string `json:"error,omitempty"` // ExceptionCode is ClickHouse's own error code when the record was refused // by the server's parser (117 unknown field — which includes a column the - // role may not write, 27 unparseable value, 6 out of range). Absent for a + // role may not write, 27 unparseable value, 41 a bad DateTime; an + // out-of-range integer wraps rather than refusing). Absent for a // gateway rejection — a failed check clause is our verdict, not // ClickHouse's, and must not be dressed as one. ExceptionCode int `json:"exception_code,omitempty"` @@ -375,8 +376,9 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { records = n } // Otherwise a single-object body is one record (concatenated objects after - // it are ignored, as they always have been — declare NDJSON to batch them, - // #561), and a line-framed body has at least the record its first byte + // it are neither answered nor published, as they always have been — declare + // NDJSON to batch them, #561; chtypes still parses them, so one cut off + // mid-record can turn the answer into a decline), and a line-framed body has at least the record its first byte // starts. The real count is chtypes' own, taken once it has answered. guard := h.policyCheckGuard(ctx, table, role, schema, perms) diff --git a/internal/api/query.go b/internal/api/query.go index 0c5af037..b622e4f9 100644 --- a/internal/api/query.go +++ b/internal/api/query.go @@ -32,11 +32,9 @@ import ( // Why a proxy instead of clickhouse-go's native Query/Exec: // - ClickHouse classifies statements natively, so any single statement // (arbitrary DDL/DML verbs, current and future) and inline FORMAT -// directives all just work without WaveHouse-side parsing. -// Multi-statement input (`SELECT 1; TRUNCATE t`) also works when -// the upstream ClickHouse has multi-query enabled, which is the -// default in recent versions; older or restrictively-configured -// servers may reject the second statement with a clear error. +// directives all just work without WaveHouse-side parsing. The HTTP +// interface takes one statement per request: `SELECT 1; TRUNCATE t` is +// refused with code 62 (measured on 26.8.15.10). // - There is no IsMutation heuristic to maintain — no leading-verb table, // no comment stripper, no CTE-aware paren scanner, no class of bug // where a future ClickHouse verb routes the wrong way. @@ -295,9 +293,10 @@ func (h *QueryHandler) Handle(w http.ResponseWriter, r *http.Request) { } if resp.StatusCode != http.StatusOK { - // ClickHouse answers most errors with HTTP 500 — bad SQL, a missing - // grant, an unknown table alike — so the status says nothing; the - // exception code it sends with it does (#403). The message is + // The status ClickHouse sends does not say which kind of failure it + // was (on 26.8 a syntax error is 400, an unknown table 404, a + // TIMEOUT_EXCEEDED 408); the exception code it sends with it does + // (#403). The message is // ClickHouse's own text, verbatim. chErr := chconn.NewHTTPError(&http.Response{StatusCode: resp.StatusCode, Header: resp.Header, Body: io.NopCloser(bytes.NewReader(body))}) msg := strings.TrimSpace(string(body)) diff --git a/internal/auth/auth_test.go b/internal/auth/auth_test.go index 8e2d7fdc..d4bc158b 100644 --- a/internal/auth/auth_test.go +++ b/internal/auth/auth_test.go @@ -189,9 +189,9 @@ func TestMiddleware_LargeIntegerClaim_ExactThroughPolicy(t *testing.T) { // TestMiddleware_NumericClaimSpelling_BindsCanonically: json.Number keeps the // token's literal spelling, so without normalization the bound filter value -// would depend on how the IdP spelled the number — and a numeric ClickHouse -// column rejects '1.0'/'1e3' as a TYPE_MISMATCH error on every query for that -// role. The claims ride a real signed token (a json.Number claim value +// would depend on how the IdP spelled the number — and an integer column +// reads only the canonical spelling, so '1.0'/'1e3' would match nothing for +// that role. The claims ride a real signed token (a json.Number claim value // marshals verbatim into the payload) so the exact parser configuration is // what's under test, per the note on the large-integer test above. func TestMiddleware_NumericClaimSpelling_BindsCanonically(t *testing.T) { diff --git a/internal/policy/canonical.go b/internal/policy/canonical.go index 91b78b5b..2689cd0c 100644 --- a/internal/policy/canonical.go +++ b/internal/policy/canonical.go @@ -33,8 +33,9 @@ const maxCanonicalDigits = 100 // every row; the one legitimate structured shape, a bare-claim _in array, is // unpacked by resolveInValues before its elements reach here. A json.Number // (jwt.WithJSONNumber on claims) binds in canonical decimal form, not the -// token's spelling: "1", "1.0", and "1e3" are one JSON value, and a numeric -// ClickHouse column rejects '1.0'/'1e3' as a per-query TYPE_MISMATCH error. The +// token's spelling: "1", "1.0", and "1e3" are one JSON value, and an integer +// column reads only the canonical spelling ('1.0'/'1e3' match nothing through +// the strict cast, and are TYPE_MISMATCH compared directly). The // canonical form is exact at every width and precision — integer literals via // big.Int, fractions and exponents via canonicalDecimal, never a float64 // round-trip that could bind a value the token doesn't carry ("1e-400" fails diff --git a/internal/policy/policy_test.go b/internal/policy/policy_test.go index c7d5a677..39615394 100644 --- a/internal/policy/policy_test.go +++ b/internal/policy/policy_test.go @@ -441,8 +441,8 @@ func TestResolveTemplate(t *testing.T) { {"boolean claim binds", "{{ jwt.is_admin }}", "true", true}, {"large integer claim binds exactly", "{{ jwt.big }}", "12345678901234567890", true}, // Numeric claims bind in canonical decimal form, not the token's - // spelling — "1.0"/"1e3" error as TYPE_MISMATCH against a numeric - // column if bound verbatim. A magnitude only JSON can hold fails + // spelling — bound verbatim, "1.0"/"1e3" would match nothing on an + // integer column, which reads only the canonical spelling. A magnitude only JSON can hold fails // closed like any other unresolvable claim. {"float spelling binds canonically", "{{ jwt.price }}", "1", true}, {"exponent spelling binds canonically", "{{ jwt.exp3 }}", "1000", true}, diff --git a/internal/query/bind.go b/internal/query/bind.go index 906b2fbb..297e0f08 100644 --- a/internal/query/bind.go +++ b/internal/query/bind.go @@ -243,9 +243,10 @@ func sqlString(s string) string { // - a value on a Date or DateTime column binds the same way and expands to // the parse its conversion names, so ClickHouse reads the caller's own // spelling — an RFC 3339 instant, or a zone-less time in the column's -// zone — instead of Go rewriting it. Compared directly, ClickHouse -// refuses RFC 3339 on DateTime and DateTime64 (TYPE_MISMATCH on -// 24.8.14.39 and 26.8.15.10). +// zone — instead of Go rewriting it. Compared directly, RFC 3339 is +// read only under cast_string_to_date_time_mode=best_effort: the +// default on 26.8.15.10, but 24.8.14.39 refused it on DateTime and +// DateTime64 (TYPE_MISMATCH), as a server set to basic still does. // - a policy claim on an integer column (chsql.IntParam) expands to // chsql.StrictInt over its one {pN:String}, because the plain form wraps // a value at or past 2^64. diff --git a/internal/stream/hub_test.go b/internal/stream/hub_test.go index 27038cf9..8c1b16e8 100644 --- a/internal/stream/hub_test.go +++ b/internal/stream/hub_test.go @@ -1112,8 +1112,8 @@ func TestHub_RowFilter_BigIntegerExact(t *testing.T) { } // TestHub_RowFilter_TimestampInstantMatch: policy authors write the zone-less -// spelling the query path wants, while the wire carries ClickHouse's RFC 3339 -// rendering. The filter compares them as instants because the row is parsed +// spelling every server's query path reads alike, while the wire carries +// ClickHouse's RFC 3339 rendering. The filter compares them as instants because the row is parsed // into the column's real storage before the predicate runs, so any spelling of // the same instant matches; an operand the parser can't read withholds the row // rather than guessing at it. @@ -1123,7 +1123,7 @@ func TestHub_RowFilter_TimestampInstantMatch(t *testing.T) { Tables: map[string]policy.TablePolicy{ "clicks": { // Zone-less constant, read in the column's zone (UTC here) on both - // surfaces — the one spelling that works for the query path's SQL too. + // surfaces, whatever the server's cast_string_to_date_time_mode. "viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"created_at": {Eq: new("2026-06-21 04:00:00")}}}}, }, }, diff --git a/internal/stream/roweval.go b/internal/stream/roweval.go index d8784a43..3722956a 100644 --- a/internal/stream/roweval.go +++ b/internal/stream/roweval.go @@ -24,8 +24,9 @@ const ( // ReasonError: ClickHouse evaluated the predicate over this row and raised. // Either side of the comparison can cause it, and both are the policy // author's to fix: a stored value the expression cannot read, or a filter - // constant the column's type cannot read (a claim rendering as "abc" or - // "1.5" against a numeric column answers code 53 per row). + // constant the column's type cannot read (a claim rendering as "abc" + // against a Decimal column answers code 53 per row, against a Float one + // 72; an integer column's strict cast answers false instead). ReasonError = typelayer.ReasonError // ReasonDecline: no verdict was reached. The expression would not compile // for this generation, chtypes would not answer for the row (one that does diff --git a/internal/typelayer/filter.go b/internal/typelayer/filter.go index e57e8ae2..5de3703a 100644 --- a/internal/typelayer/filter.go +++ b/internal/typelayer/filter.go @@ -186,8 +186,8 @@ func verdictBool(v chtypes.Verdict) (bool, string) { // ClickHouse's own comparison-time coercion (a typed parameter disagreed with // the server on UInt8, Int64 and Float32 columns; the String binding matched // it on every column family measured): a spelling the column cannot read -// (`abc` on a Float32 column) is the server's own code 53 at evaluation, -// which withholds the row. A bare String binding still wraps an +// (`abc` on a Float32 column, the server's own code 72; on a Decimal, 53) +// errors at evaluation, which withholds the row. A bare String binding still wraps an // integer value at or past 2^64 before comparing (and a 128/256-bit column at // its own width), so on an integer column the parameter is compared through // chsql.StrictInt instead: a claim that is not the canonical spelling of a diff --git a/internal/typelayer/filter_test.go b/internal/typelayer/filter_test.go index 8639ab6f..a0f23331 100644 --- a/internal/typelayer/filter_test.go +++ b/internal/typelayer/filter_test.go @@ -117,8 +117,8 @@ func TestVisible_StringBindingAcrossColumnFamilies(t *testing.T) { // a spelling that is not its canonical form, matches nothing on EVERY operator // — `!=` included — and is answered false rather than thrown. On a // non-integer column the String binding is unchanged: a spelling the column's -// reader refuses is still the server's own code 53, which withholds as an -// error. +// reader refuses is still the server's own error (code 72 on this Float32 +// column), which withholds as an error. func TestVisible_HostileSpellingsMatchNothing(t *testing.T) { _, row := parsedRow(t) diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go index c3351bb8..d79b5583 100644 --- a/internal/typelayer/ingest.go +++ b/internal/typelayer/ingest.go @@ -63,7 +63,7 @@ func parseSettings(format Format, opts IngestOptions) (settings map[string]strin type RowVerdict struct { Accepted bool // Code is ClickHouse's own error code when the record was refused (27, 117, - // 6 …) and 0 otherwise. Declined verdicts carry no code: chtypes did not + // 41 …) and 0 otherwise. Declined verdicts carry no code: chtypes did not // answer, so there is nothing to attribute to the data. Code int Message string From 4d35acfc15babfd46605baf8e7deccafdd757ad8 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 10:20:50 -0400 Subject: [PATCH 49/70] docs: correct ingest codes, timestamp claims and stale mechanisms Measured on ClickHouse 26.8.15.10 and the 26.8 chtypes artifact: - Per-record ingest codes are 27/26/33 plus the type's own (41, 38, 69, 72, 376, 467, ...); code 6 never occurs and out-of-range integers wrap. A quoted "true" into a Bool is refused (467). Truncated bodies, a bad element in a single-line array declared NDJSON, and an NDJSON line that absorbs the next are documented. - RFC 3339 policy claims match on DateTime/DateTime64 under 26.8's best_effort default and are refused before 26.5; a Date column takes YYYY-MM-DD only. Decimal/Float claims a column cannot read fail the whole /v1/query read with 400 clickhouse.rejected. - /v1/ops/query takes one statement per request; its 413 and mid-stream exception behaviour are documented. - Error tables gain missing rows (missing table, 401, 404, schema not loaded, header refusal, no columns readable, 422 in the SDK table). - Deployment: fetch step before a source build, CHTYPES_AUTOFETCH, mounting another line's artifact, probes staying green, the upgrade audit, one time zone per line in nested deployments. - Development and AGENTS: the chtypes artifact as a make ci prerequisite, the self-managed E2E recipe, OTel export under make dev, project trees, doc-sync rows; CHANGELOG Unreleased entries corrected (released sections untouched). Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 34 ++++---- CHANGELOG.md | 36 ++++----- CONTRIBUTING.md | 6 +- README.md | 12 +-- SECURITY.md | 6 +- docs/src/content/docs/access-control.mdx | 22 +++--- docs/src/content/docs/api.md | 82 +++++++++++--------- docs/src/content/docs/architecture.md | 54 +++++++------ docs/src/content/docs/claude-code.md | 4 +- docs/src/content/docs/configuration.mdx | 10 +-- docs/src/content/docs/deployment.md | 29 ++++--- docs/src/content/docs/development.md | 44 ++++++----- docs/src/content/docs/durability.md | 2 +- docs/src/content/docs/getting-started.md | 8 +- docs/src/content/docs/ingest-pipeline.md | 2 +- docs/src/content/docs/pipes.mdx | 7 +- docs/src/content/docs/reverse-proxy.mdx | 4 +- docs/src/content/docs/sdk/reference.md | 3 +- docs/src/content/docs/sdk/streaming.md | 6 +- docs/src/content/docs/settings-directory.mdx | 12 +-- docs/src/content/docs/why-wavehouse.md | 12 +-- 21 files changed, 215 insertions(+), 180 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index b3d500e8..3e89c49c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -28,7 +28,7 @@ One binary: Twenty-one internal packages under `internal/` (plus `internal/testutil/` for shared test helpers): -- **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers. On the ingest path, `content_type.go` is the Content-Type → format table (and the RFC 9110 resolution rules behind the `415`) and `ingest_framing.go` is the only code that reads body bytes itself — the array re-frame, the record count, and the positional dedupe-id read. `clickhouse_http.go` runs structured queries and pipes over each tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, so ClickHouse renders every row; `ch_errors.go` (`writeCHError`) is the one mapping from a failed ClickHouse query to status, `code` and `retryable` (`writeCHWriteError` for a write pipe: never retryable, no `Retry-After`, since the write may have run) +- **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers. On the ingest path, `content_type.go` is the Content-Type → format table (and the RFC 9110 resolution rules behind the `415`) and `ingest_framing.go` is the only code in `internal/api` that reads body bytes itself — the array re-frame, the record count, and the positional dedupe-id read. `clickhouse_http.go` runs structured queries and pipes over each tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, so ClickHouse renders every row; `ch_errors.go` (`writeCHError`) is the one mapping from a failed ClickHouse query to status, `code` and `retryable` (`writeCHWriteError` for a write pipe: never retryable, no `Retry-After`, since the write may have run) - **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive` for the one setting folded over every tenant served, `gapWindows` handing the sweeper each tenant's own gap window (a rejected tenant's as its folder last had it, unbounded for one rejected since boot) and the `mq.max_bytes_gb` reconcile each served tenant's byte budget, and `defaultPolicy` for the one setting that still follows tenant `0`, a flat directory's ops-gate admin role; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, the same hook's `Hub.Prune` ends the open streams of a tenant no longer served, and `wireCache`'s hook drops, through `LocalCache.Prune`, the cache version index of a tenant no longer served ([#262](https://github.com/Wave-RF/WaveHouse/issues/262))), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `New` wires only what the process's `roles` need (discovery, dedupe, auth verifiers, the hub bridge and keepalive per API process; the ingest worker per ingest process; the sweeper under its lease through `elected`, on the embedded MQ only); a process without `api` serves `api.NewOpsRouter` — probes, `/version`, metrics, and the settings reload behind the operator key alone. `cmd/wavehouse` and `tests/integration` both boot through it - **`auth/`** — JWT auth middleware: HMAC **or** JWKS verification with `alg` pinned to the active verifier, role extraction from a configurable claim path; always runs, never rejects (bad token → empty role + stashed reason). One verifier per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9): `Authenticator` keys them by `tenant.ID` — the request store's `settings.Store.Tenant()`, through an injected `TenantSource`; `tenant.Default` on the tenant-exempt routes — built from each tenant's `auth` block by `Reconfigure`, dropped by `Prune` once the tenant stops being served (rejected or removed), released by `Close`; the secrets (`Config`) are boot-level and shared. A JWKS key set is fetched off the boot and reload paths: until one has been stored the verifier is pending and a token-bearing request gets `503` + `Retry-After` from `api.refuseUnverifiable` (`auth.ErrVerifierPending`), never a `default_role` evaluation; refresh is library-managed (Eric, 2026-09-22), response capped at 1 MiB; the operator key's admin role is the request tenant's - **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index), and `RedisCache`, the Redis-compatible shared backend (random version tokens under the tenant's hash tag, one-round-trip lookups, bypass on failure behind a circuit breaker, deferred invalidations retried; selected by `cache.backend: redis`, configured by the boot config's `cache.redis` block — [#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every key carries the tenant (in `RedisCache`, after the key prefix: `:{}:…` for a version token, `:q::…` for a value); in `LocalCache` and the version index it leads ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for the caller's query key and its singleflight, escaped whole as the lead field of the stored key `|.||…`, where each raw table and scope name is escaped by `keyenc` (a `Namespace` carries them raw, so no caller escapes); the index holds a version per tenant, per (tenant, table) and per (tenant, table, scope), keyed by raw name and bumped in place (one entry per live namespace however often it is bumped, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)) — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` drops the tenant's index so its next key gets a process-unique generation, orphaning its every cached result in one step, pipe results included (no insert reaches a pipe result until [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); `Lookup` returns a `Snapshot` of the versions it read, taken before the handler chooses any input a bump invalidates — the tenant's connection included — and `Set` files the fill under it, so a write landing mid-query, or a reload moving the tenant to another address or database after the request took its connection, orphans the fill ([#382](https://github.com/Wave-RF/WaveHouse/issues/382)), and every backend runs the conformance suite `internal/testutil/cachetest`; the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the whole cache of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) @@ -47,7 +47,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes - **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row via `typelayer`, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`typelayer/`** — the only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns (a denied column stays as `MATERIALIZED` of its default, so naming it is code 117), plus a `DEFAULT ''` per `_eq` check column, and any `EPHEMERAL` column the role may write that a `DEFAULT` reads and no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` expression reads — which is how column policy and auto-inject are answered with no Go-side record inspection; a role shape's handle pool grows to `min(GOMAXPROCS, 4)` under load. A role schema that cannot compile is `500` with `retryable:false`. Test helpers live in `internal/typelayer/typelayertest` (`TestEngine`, `SkipWithoutArtifact`). `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) +- **`typelayer/`** — the only package (with its test helper `typelayertest`) that imports `github.com/wave-rf/chtypes/go/chtypes`, by convention: no depguard rule enforces it. One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns (a denied column stays as `MATERIALIZED` of its default, so naming it is code 117), plus a `DEFAULT ''` per `_eq` check column, and any `EPHEMERAL` column the role may write that a `DEFAULT` reads and no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` expression reads — which is how column policy and auto-inject are answered with no Go-side record inspection; a role shape's handle pool grows to `min(GOMAXPROCS, 4)` under load. A role schema that cannot compile is `500` with `retryable:false`. Test helpers live in `internal/typelayer/typelayertest` (`TestEngine`, `SkipWithoutArtifact`). `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) - **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -61,7 +61,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 5. **Per-tenant-table batching** — the worker groups events by tenant table (the tenant read off each message's `mq.Topic`), so one INSERT never mixes tenants and a batch invalidates its own tenant's cache namespaces; then it splits each batch by column list (`groupByColumns`), emitting one `INSERT INTO … (cols) FORMAT JSONCompactEachRow` per distinct list so a schema change mid-stream can't corrupt a statement. Each tenant table's batch is independent. 6. **Dead Letter Queue** — batch inserts ClickHouse **rejects** (isolated row by row; `chconn.Classify` == `Rejected` — a multi-row batch refused for its size, `chconn.Splittable`, is split row by row too) publish to the tenant's own dead-letter queue (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). A ClickHouse that cannot take the insert — unavailable, denied, or no verdict — never dead-letters a row, not even mid-isolation: the rows go back to the MQ with a delayed nak under a per-pool backoff — per table for a failure of one table (`chconn.TableScoped`: read-only, too many parts or mutations, a missing grant; `internal/ingest/backoff.go`), counted by `wavehouse_ingest_retries_total`. No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format`, or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. 7. **Auth: always on, fail-loud, decoupled from authz (security)** — the JWT middleware always runs (no `auth.enabled`/`dev_mode` flag); it verifies with HMAC **or** JWKS (not both), with accepted `alg` pinned to the active verifier and checked before any key is used (rejects `alg:none` and cross-family confusion). No/invalid/expired token → empty role → policy `default_role`, with the bad-token reason stashed so a denying gate returns a loud `401`, not a bare `403`; the one token outcome that never reaches `default_role` is a verifier still fetching its JWKS (`auth.ErrVerifierPending` → `503` + `Retry-After`, `api.refuseUnverifiable`). Elevated access needs a valid granted role. **Sanctioned exception:** a configured non-JWT operator key (`auth.operator_key`; presented via `Authorization: Operator ` or the `X-Operator-Key` alias) deliberately couples authN+authZ — a constant-time match authorizes a full-access platform operator (stamps the admin role plus an operator bit) independent of the verifier (see #11). Detail: architecture.md § `api/` + `internal/auth`; see also #11, §Security Considerations. -8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant and table; claims are two-phase, one call per phase per window of up to 256 records — `Reserve` → publish (under the id's idempotency key) → `Commit`, or `Release` when the publish definitely failed, while one whose outcome is unknown is left to lapse; a store that cannot answer is a `503` + `Retry-After`; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key and `dedupe.retention` how long a committed id stays a duplicate (`"0"` = forever, else at least the queue's two-minute duplicate window), both overridable per table. +8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant and table; claims are two-phase, one call per phase per window of up to 256 records — `Reserve` → publish (under the id's idempotency key) → `Commit`, or `Release` when the publish definitely failed, while one whose outcome is unknown is left to lapse; a store that cannot answer is a `503` + `Retry-After`; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the column whose rendered value is the id and `dedupe.retention` how long a committed id stays a duplicate (`"0"` = forever, else at least `2m`, the embedded queue's duplicate window), both overridable per table. 9. **Singleflight** — the cached read handlers coalesce concurrent misses (`x/sync/singleflight`) under the tenant-led cache key to prevent cache stampede, per tenant. 10. **Active Sweeper** — purges NATS messages that are both ACKed (written to CH) and older than the gap window; SSE gap-fill uses `DeliverByStartTime`, no in-process ring buffer. It runs only in the process holding the `sweeper` lease (`coord.RunElected`); that lease is not fenced: an overlap cannot lose ClickHouse data, since every sweep stops at the consumer's ack floor; it can only trim SSE replay history, and only when the two holders' settings views differ (one still reading a shorter `stream.gap_window_minutes`, or missing a tenant, after a reload the other has applied) — which fencing would not prevent either. Anything that does need exclusivity must check the term's `Token`. 11. **Hasura-style access control: fail-closed (security)** — `policy.IsAdmin` (role == `admin_role`, **exact case-sensitive**, default `"admin"`) is the single admin check, shared by `Evaluate`/`ResolveRole`/`Validate`/the `/v1/ops` gate/`RoleAllowed`. Empty/absent role matches nothing (no `"*"` wildcard); `Validate` rejects empty role keys; a `nil` policy (deleted) denies **everyone incl. admin** via a role — a total lockout for token-based callers, so recovery is writing `policies.json` and reloading, never an implicit admin grant (**exception:** the operator key's `auth.IsOperator` bit passes the `/v1/ops` gate even under a `nil` policy — a deliberate break-glass that can `POST /v1/ops/settings/reload` over HTTP, see #7). Over a nested settings directory the `/v1/ops` gate reads no policy at all — those routes reach every tenant, so the operator key alone passes and an admin-role token gets `403`; `api.NewRouter` decides that from the registry's shape, not from what was wired. `default_role` is the one sanctioned roleless exception (`ResolveRole` maps empty → it pre-eval); `default_role == admin_role` is permitted but dev-only and loudly warned (`policy.DefaultRoleGrantsAdmin`). Preserve when touching `internal/policy` (policy twin of #13; see #159). Detail: architecture.md § `policy/`. @@ -74,7 +74,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention; `/readyz` pings every open ClickHouse pool at once and is ready at the first answer, 503 naming each when none answers); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. 19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), as RFC 3339 in UTC at the column's scale (e.g. `"2026-06-21T04:00:00.123Z"`), because `date_time_output_format=iso` is pinned on both surfaces — the ingest export settings (`internal/typelayer/ingest.go`) and the read path's `chReadSettingsFixed` (`internal/api/clickhouse_http.go`), which must change together with a `chRendering` bump — so there is no separate WaveHouse rewrite step to keep in sync and live and query reads can't drift on spelling *or* instant (#372). The column's or server's zone affects only how zone-less *input* is read, never the rendered spelling. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. 20. **Sealed MQ boundary** — only `internal/mq` imports NATS/JetStream (`github.com/nats-io/…`), enforced by the `depguard` rule in `.golangci.yml`, so `make lint` fails on a leak in every package it builds (the `integration`-tagged files under `tests/` are outside lint's build context — keep them clean by convention, through `mq.Broker`). A test outside `internal/mq` that needs a real NATS server goes through `internal/mq/natstest`, which stands one up from the shipped `deployments/nats` files and hands back a URL and passwords, never a NATS type. The boundary is semantic as well: everything else addresses events by `mq.Topic` and states intent through mq-owned interfaces (`Publisher`, `Consumer`, `DeadLetterer`, `Purger`, `Replayer`, …), and never builds a subject, names a stream, or reasons in sequences — so a subject, stream, or broker change lands in one package ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 4; story 5's tenant token landed there alone — `Topic.Tenant`, first in every subject). Don't add a raw accessor (`JetStream()`, `NatsConn()`, `GetServer()`) back, and don't hand-build `"ingest."`/`"dlq."` subjects outside `internal/mq` — widen the mq surface with an intent-level method instead. -21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503` (generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds every row of a role that has a row `filter` with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and, on Linux, glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. +21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` (with its test helper `typelayertest`) is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503` (generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds every row of a role that has a row `filter` with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and, on Linux, glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. ## Code Conventions @@ -124,18 +124,20 @@ Tooling notes (the non-obvious bits `make help` won't tell you): - **GNU Make 4+** required (uses `--output-sync=target`); macOS BSD Make 3.81 won't parse it. Full setup: `docs/src/content/docs/development.md` § Prerequisites. - **Lint split**: Biome owns JS/TS/JSON, markdownlint owns Markdown *and MDX* style — including two repo-local rules, WH001 (no hard-wrapped prose) and WH002 (MDX fence beside a JSX tag) in `scripts/markdownlint-rules/` — misspell owns spelling (all under `make lint`/`make fix`); accuracy/clarity/doc-sync is the `docs-reviewer` gate (§Docs review). See §Markdown authoring rules. - **Worktrunk** (`wt`, `.config/wt.toml`): `wt switch --create` seeds `.bin/` + `node_modules/` from main, then runs `make tools`. +- **chtypes artifact**: `make tools` does not fetch it; run `scripts/fetch-chtypes.sh` once per machine (it lands in `~/.cache/chtypes/artifacts/abi6/-`). An API process, the integration suite and the E2E server refuse to boot without it. ## Testing Conventions - **Table-driven tests**: Use `tests := []struct{ name string; ... }` with `t.Run(tt.name, ...)` for test cases. - **Shared mocks in `internal/testutil/`**: Use `MockPublisher` (records `Publish` and `DeadLetter`), `MockCache`, `MockDeduplicator`, `MockSubscriber`, `MockMessage`, `MockPurger`, `MockDeadLetterStats` instead of creating ad-hoc mocks. See `testutil/mocks.go`. - **JWT helpers**: Use `testutil.MakeJWT(t, claims)` and `testutil.MakeExpiredJWT(t, claims)` for auth tests. See `testutil/jwt.go`. -- **Schema helpers**: Use `testutil.NewTestSchemaRegistry(t, tables)` for schema-aware tests — it builds the registry through the real discovery path (`Refresh` against a mock ClickHouse connection), so the derived fields are computed exactly as in production. +- **Schema helpers**: Use `testutil.NewTestSchemaRegistry(t, tables)` for schema-aware tests — it builds the registry through the real discovery path (`Refresh` against a mock ClickHouse connection), so the derived fields are computed exactly as in production. It binds no type-layer engine. +- **Type-layer helpers**: a test that parses ingest bodies or evaluates row filters takes its engine from `typelayertest.TestEngine(t, tables...)` (`internal/typelayer/typelayertest`). It skips without the chtypes artifact (install it with `scripts/fetch-chtypes.sh`); set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1`, as CI does, to fail instead. - **Cache backends**: every `cache.Cache` backend runs `cachetest.Run` (`internal/testutil/cachetest`), the backend-agnostic conformance suite; a behavior the contract promises goes there, not in one backend's tests. `RedisCache` runs it from `internal/cache/redis_integration_test.go` (`//go:build integration`, pinned Redis, Valkey, Dragonfly and Redis Cluster containers), which `make test-integration` includes. - **Write classifier cases**: `api.IsMutation`'s cases live in `internal/testutil/mutationtest`, shared by its unit test and `tests/integration/ismutation_test.go`, which checks each against ClickHouse's own parser; add a case there, not to either test. - **Policy helpers**: Use `policy.Static(p)` for a fixed `policy.Source` in tests. - **Pipes helpers**: Use `pipes.Static(queries...)` for a fixed `pipes.Source` in tests. -- **Response assertions**: Use `testutil.AssertJSONResponse(t, rec, status, expected)` and `testutil.AssertJSONContains(t, rec, status, substring)`. +- **Response assertions**: Use `testutil.AssertJSONResponse(t, rec, status, expected)` and `testutil.AssertJSONContains(t, rec, status, map[string]any{…})` (checks those keys' values). - **Coverage target**: 80% project-wide (CI enforces `threshold.total` in `.testcoverage.yml` against the merged unit + integration + e2e profile). Per-suite minima also enforced: unit 80%, integration 20%, e2e 60%, sdk 50%. Aim for 80%+ on new code. Coverage is published as PR comments via GitHub Code Quality — see `.github/workflows/README.md` "Coverage publishing"; the gate is unchanged. - **Every new function should have corresponding test cases.** Run `make lint` and `make test` before considering work complete. - **E2E tests via SDK**: The TypeScript SDK is the primary E2E test harness. Tests in `tests/e2e/sdk/` exercise the full pipeline (ingest → ClickHouse → query) and simultaneously validate backend behavior and SDK correctness. Use `make test-e2e` to run. Add new E2E scenarios as `tests/e2e/sdk/*.test.ts` files using helpers from `tests/e2e/sdk/helpers.ts`. @@ -155,7 +157,7 @@ If `make ci` passes locally, your commit has crossed the same gates CI will run ### Running `make ci` (for agents) -`make ci` is **self-contained**: the integration suite (`tests/integration/`) and the E2E orchestrator (`scripts/orchestrator/`) each boot ClickHouse and a Redis via **testcontainers on random host ports** (the integration suite also dynamodb-local), and the shared cache backend's integration tests (`internal/cache/`) start their own Redis, Valkey, Dragonfly and one-node Redis Cluster containers the same way. The only prerequisite is a running **Docker daemon** — do **not** `make deps-up` or start ClickHouse first (`deps-up` is for `make dev` only). +`make ci` is **self-contained**: the integration suite (`tests/integration/`) and the E2E orchestrator (`scripts/orchestrator/`) each boot ClickHouse and a Redis via **testcontainers on random host ports** (the integration suite also dynamodb-local), and the shared cache backend's integration tests (`internal/cache/`) start their own Redis, Valkey, Dragonfly and one-node Redis Cluster containers the same way. The prerequisites are a running **Docker daemon** and the chtypes artifact — run `scripts/fetch-chtypes.sh` once per machine (`make tools` does not fetch it); without it the integration suite and the E2E server refuse to boot and the engine-dependent unit tests skip (CI sets `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` to make them fail). Do **not** `make deps-up` or start ClickHouse first (`deps-up` is for `make dev` only). Run it via the **background Bash tool** (`run_in_background: true`) and wait for the completion notification; the harness re-invokes you on exit, so polling the log with `tail` only burns context: @@ -318,7 +320,8 @@ Every code change should update the corresponding docs in the same PR. A code ch | Add/modify API endpoint | `docs/src/content/docs/api.md`, `README.md` (if user-facing) | | Add/modify boot config option | `docs/src/content/docs/configuration.mdx`, `config.yaml`, `deployments/compose/*` env blocks, `docs/src/content/docs/deployment.md` | | Add/modify a settings-directory key (`config.json`, `roles.json`, `policies.json`, `pipes.json`) | `docs/src/content/docs/settings-directory.mdx` (table **and** the example block), `internal/settings/seed/*.json`, `deployments/compose/settings/*.json`, `tests/e2e/fixtures/settings/*.json`, `config.yaml` (the `settings:` comment enumerates the keys) | -| Change architecture / add a package | `docs/src/content/docs/architecture.md`, `AGENTS.md` | +| Change architecture / add a package | `docs/src/content/docs/architecture.md`, `AGENTS.md` (§File Structure), `docs/src/content/docs/development.md` (Project Structure) | +| Bump the ClickHouse line, the chtypes SDK, or `chtypes.lock` | `scripts/fetch-chtypes.sh` (`LOCK_LINES`, `CHTYPES_SDK_VERSION` ↔ `go.mod`), `chtypes.lock`, the `clickhouse-server` tags in `deployments/compose/*`, `tests/integration/setup_test.go`, `scripts/orchestrator/main.go`, `typelayertest.TestServerVersion`, `.github/actions/setup-env` (the `abi6` cache key), `docs/src/content/docs/deployment.md` (§chtypes artifacts), `development.md`, `configuration.mdx`, `README.md` — then grep for the old version | | Change ingest / event format | `docs/src/content/docs/api.md`, `docs/src/content/docs/deployment.md` (CH schema) | | Change deployment / Docker | `docs/src/content/docs/deployment.md`, compose files | | Change build / test process | `docs/src/content/docs/development.md`, `Makefile` | @@ -413,7 +416,7 @@ Internal-only backend changes (middleware refactors, observability internals, de 1. Create the package under `internal/`. 2. Define an interface if there will be multiple implementations. 3. Wire it into `internal/app/wire.go` as a component (what it opens, what it loops, what it releases). -4. Document in `docs/src/content/docs/architecture.md`. +4. Document in `docs/src/content/docs/architecture.md`, and add it to `AGENTS.md` §File Structure and the Project Structure tree in `docs/src/content/docs/development.md`. 5. **Add a matching `area/` repo label** (e.g. `area/foo` for `internal/foo/`) so issues can be routed to it during triage, and add the path → label mapping to `.github/labeler.yml` so PRs touching the package get auto-labeled. Issue triage is manual (see §Repository Automation); only the PR-side labeling is automated. ### Writing tests @@ -449,18 +452,21 @@ internal/policy/ → Access control policies (types, evaluation, Source) internal/query/ → Structured query AST + SQL builder internal/settings/ → Settings directory (validate, adopted snapshot + reload, watcher, embedded seed) internal/stream/ → SSE fan-out (event Hub: project once per role, Subscriber outbound queue, Bucket fan-out, keepalive Heartbeater wheel) -internal/typelayer/ → In-process ClickHouse parser (chtypes): ingest validation/coercion, insert checks + row-level-security compilation (typelayertest/ builds a test engine from the locked artifact) internal/tenant/ → Tenant id (type, grammar, reserved default, request header name) internal/testutil/ → Shared test helpers (mocks, JWT + schema helpers; logtest/ captures or silences the default logger; cachetest/ is the conformance suite every cache.Cache backend runs; mutationtest/ holds the shared write-classifier cases; storedir/ is the embedded broker's store directory in tests, removed once late consumer-state writes land) +internal/typelayer/ → In-process ClickHouse parser (chtypes): ingest validation/coercion, insert checks + row-level-security compilation (typelayertest/ builds a test engine from the locked artifact) tests/ → Integration & E2E tests -tests/integration/ → Go integration tests (//go:build integration; ClickHouse testcontainer, Redis for shared_cache_test.go, and NATS for the `mq.backend: nats` end-to-end test). A package tested against its own external server keeps them beside it: internal/cache/redis_integration_test.go (Redis, Valkey, Dragonfly, Redis Cluster testcontainers). `make test-integration` also runs `internal/mq/natsspike` (nats-server semantics, under `internal/mq` for the NATS import boundary) and `internal/mq`'s integration-tagged external-NATS broker tests (`TestExternalNATS*`, `TestNewNATS*`, `TestNATSPermissions_Refuse*`, `TestLeases*`) +tests/integration/ → Go integration tests (//go:build integration; ClickHouse and dynamodb-local testcontainers, Redis for shared_cache_test.go, and NATS for the `mq.backend: nats` end-to-end test). A package tested against its own external server keeps them beside it: internal/cache/redis_integration_test.go (Redis, Valkey, Dragonfly, Redis Cluster testcontainers). `make test-integration` also runs `internal/mq/natsspike` (nats-server semantics, under `internal/mq` for the NATS import boundary), `internal/mq`'s integration-tagged external-NATS broker tests (`TestExternalNATS*`, `TestNewNATS*`, `TestNATSPermissions_Refuse*`, `TestLeases*`), and `internal/api`'s `TestIntegration_*` (the read path's filters against a ClickHouse in a non-UTC zone) tests/e2e/ → E2E test stack (scripts/orchestrator boots ClickHouse and Redis testcontainers + the wavehouse-cov binary) -tests/e2e/fixtures/ → Idempotent ClickHouse DDL scripts for test tables +tests/e2e/fixtures/ → The E2E server's boot config (config.yaml) and settings directory (settings/), copied per run by the orchestrator tests/e2e/sdk/ → E2E integration tests via TypeScript SDK (Vitest) -deployments/compose/ → Docker Compose files (standalone.yaml, dependencies.yaml) +deployments/compose/ → Docker Compose files (standalone.yaml, dependencies.yaml) + settings/ (the trial settings directory) deployments/Dockerfile → Runtime image (+ Dockerfile.goreleaser for release builds) deployments/nats/ → External NATS JetStream: nack CRs (`wavehouse mq manifests` output, golden-tested) + Helm values with WaveHouse's user permissions (test-pinned) docs/ → Project documentation +clients/ts/ → TypeScript SDK (@wavehouse/sdk) +scripts/ → E2E orchestrator, cov tool, CI logic (ci/), gate/hook helpers, fetch-chtypes.sh +chtypes.lock → Pinned chtypes artifacts per platform and ClickHouse patch .vscode/ → Workspace settings (gopls build flags, recommended extensions) ``` @@ -468,7 +474,7 @@ docs/ → Project documentation - JWT secret (or JWKS endpoint) must be cryptographically strong in production — the JWT middleware always runs (no enable flag), so token validation is the sole gate on elevated access - All `/v1/*` routes run the JWT auth middleware (always on); a request with no/invalid token falls back to the policy `default_role` -- Input JSON is validated against ClickHouse schemas before processing +- Every ingest body (JSON, NDJSON, CSV, TSV) is parsed by ClickHouse's own reader in-process (`internal/typelayer`, Key Design Decision #21) before it is acknowledged - **Column-level access control is a hard cap on every read path.** A role's `allow_columns`/`deny_columns` is enforced against *every* column a structured query references (projection, aggregations, `filters`, `group_by`, `order_by`, `time_range`) inside `query.Build`, and a `select_all` request expands to the role's allowed columns rather than `SELECT *` (an omitted projection selects nothing; `["*"]` is a literal column, not a wildcard — see Key Design Decision #12). The structured-query and live-stream paths share one decision function (`policy.IsColumnAllowed`). Don't move column checks out of the builder or special-case `SELECT *` — that reintroduces the #223 fail-open. - ClickHouse queries are passed through directly — use appropriate access controls on ClickHouse itself - **Dependency vulnerability scanning**: `govulncheck ./...` runs in CI on every push/PR. Dependabot (`.github/dependabot.yml`) opens weekly grouped PRs for outdated Go modules and GitHub Actions. diff --git a/CHANGELOG.md b/CHANGELOG.md index f0d6fe5e..f7c4f793 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -35,7 +35,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **A nested settings directory serves one tenant per folder, failing closed per tenant** (`internal/settings/{tree,registry,store,validate,watch}.go` (`tree.go` new, + tests), `internal/api/{tenant,router,pipes,settings}.go`, `internal/app/{app,wire}.go`, `cmd/wavehouse/validate.go`, `clients/ts/src/{pipes,settings,types,index}.ts`, `tests/e2e/sdk/admin.test.ts`, `docs/src/content/docs/{deployment,api,architecture}.md`, `docs/src/content/docs/{settings-directory,reverse-proxy,access-control}.mdx`, `docs/src/content/docs/sdk/{admin,pipes,reference,streaming}.md`): story 2 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files — the same findings byte for byte, the same boot refusal, the same keep-previous on a rejected reload, the same watcher and `SIGHUP`, the same `wavehouse validate` exit codes. `settings.Validate` now reads the directory's shape off its entries — any of the four file names makes it flat, otherwise a folder makes it nested — and checks either one: each tenant folder goes through the per-directory checks (now `ValidateDir`), its name through the tenant-id grammar, and its findings carry the folder (`acme/policies.json`). The shapes never mix: a flat directory rejects a folder and a nested one rejects a loose file. `settings.Open` returns the `Registry`, which now owns `Reload`, the new `ReloadTenant`, the `AfterAdopt` hooks (handed the tenants a reload adopted), and the watcher; a `Store` is a passive holder of one tenant's document. A nested directory fails closed per tenant, at boot and on reload alike: a folder with an error finding stops its tenant being served — the tenant routes answer a bare `503 {"error": "tenant settings are invalid"}`, since they resolve before authentication and the findings quote the settings — while every other tenant carries on, with no previous-snapshot fallback; a request already admitted finishes on the document it started with, and the tenant keeps one `*Store` across the rejection. A whole-directory reload mirrors the folders (a new one is served, a removed one is a `404`), and a finding about the directory itself — a loose file, an unreadable directory, a changed shape — refuses boot and rejects a reload whole, leaving every tenant as it was. A nested directory gets no watcher; `SIGHUP` reloads the whole tree in both shapes. `POST /v1/ops/settings/reload` and the admin pipe reads (`GET /v1/ops/pipes[/{name}]`) take an optional `?tenant=`, parsed strictly — a query string that does not parse, or an empty, repeated, or malformed `tenant`, is a `400`, never a read of the default tenant — with `404` for an unknown tenant; absent means the whole directory on the reload and tenant `0` on the reads. The reload response keeps its shape: over a nested directory `adopted: false` with a `422` can mean adopted in part, the rejected folders being the ones with an error among their `findings`. Over a nested directory the `/v1/ops/*` gate admits the operator key alone — those routes reach every tenant — and an admin-role token gets `403`; `api.NewRouter` decides that from the registry's shape, whatever policy source was wired. Booting a nested directory with no `auth.operator_key` therefore leaves `SIGHUP` as the only reload, and boot warns about it; a whole-directory reload that drops a tenant names it in the log; and `Registry.Watch` refuses a nested directory itself. The resources a process still has one of (ClickHouse connection, dedupe store, MQ byte budget, auth verifier) follow the settings tenant `0` last adopted, their reload hooks running only when tenant `0` is adopted, so another tenant's reload never moves them and a `0` folder that a reload rejects or removes leaves all of them as they were; a nested directory without a `0` folder boots with them unconfigured (no ClickHouse address, `/livez` degraded) and says so once at boot, and the async paths (ingest worker, sweeper, stream hub, schema refresh) stay wired to tenant `0` until story 5. The SSE keepalive wheel is the one shared resource that weighs every tenant: it runs at the shortest `stream.keepalive_interval` among the tenants being served ([#597](https://github.com/Wave-RF/WaveHouse/issues/597) tracks honoring each tenant's own). The SDK's `wh.pipes.list()`, `wh.pipes.get()`, and `wh.settings.reload()` take a `tenant` option (the new `OpsRequestOptions`) sent as `?tenant=`, since `options.headers` cannot set a query parameter. - **Requests resolve to a tenant before authentication, and the tenant is threaded through every settings read** (`internal/tenant/` (new, + tests), `internal/settings/registry.go` (new, + tests), `internal/api/tenant.go` (new, + tests), `internal/api/{router,ingest,structured_query,pipes}.go`, `internal/ingest/{worker,sweeper}.go`, `internal/stream/hub.go`, `internal/discovery/discovery.go`, `internal/app/{app,wire}.go`): story 1 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a deployment that sends no tenant header. `internal/tenant` defines the id — a validated string (letters, digits, `_`, `-`; at most 64 bytes, so it is safe as a folder name and as an MQ subject token), the reserved default `0`, and the `X-Tenant-ID` header name — and imports nothing from the rest of the repository. `api.TenantMW` runs ahead of the auth middleware on every `/v1` route outside `/v1/ops/*`: an absent or empty header is tenant `0`, a malformed id or a repeated header is a `400`, a well-formed id the new `settings.Registry` does not hold is a `404`, and the resolved `*settings.Store` rides the request context. Every answer from the middleware, the `400` and `404` included, carries `Vary: X-Tenant-ID` so a shared cache can't replay one tenant's response to another. The registry holds the one store `settings.Open` adopted, keyed `0`. Handlers read the store once and pass it down as an argument — the ingest, structured-query, and pipe getters (`PolicySource`, `DedupeSettings`, `bucketSecs`, `defaultMaxRows`, the pipes source) now take it as a parameter — and a tenant route reached without a resolved tenant answers `500` rather than fall back to one. The ingest worker, sweeper, stream hub, and schema registry are constructed with a `tenant.ID` (`tenant.Default` in `internal/app`) and their settings getters take it; a tenant the registry does not hold is logged and read as the getter's zero value, with two fail-safes — the worker's DLQ switch reads as on (an unreadable message is parked, never dropped) and the schema auto-refresh keeps its cadence rather than hand `time.NewTicker` a zero interval. The probes, `/version`, the metrics path, and `/v1/ops/*` stay tenant-exempt, with the ops tree behind the auth middleware and the admin gate exactly as before; the admin pipe reads (`GET /v1/ops/pipes[/{name}]`) serve the default tenant. `X-Tenant-ID` joins the CORS `Access-Control-Allow-Headers` list so a browser client can send it; the SDK needs no change (`options.headers`). -- **Ingest accepts CSV and TSV bodies** (`internal/api/content_type.go`, `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `docs/src/content/docs/api.md`): `Content-Type: text/csv` and `text/tab-separated-values` are read by ClickHouse's own `CSV`/`TSV` readers. `text/csv; header=present` and `text/tab-separated-values; header=present` read `CSVWithNames` / `TSVWithNames` (the first line names the columns, in any order; an omitted column takes its `DEFAULT`; an unknown or duplicate name is code 117 for the whole request). `header` is RFC 4180 §3's optional parameter, mapped three ways onto ClickHouse: `header=present` is `CSVWithNames`, `header=absent` is strictly positional (`input_format_csv_detect_header=0` / `input_format_tsv_detect_header=0`), and a bare type is ClickHouse's default reading with header auto-detection on, so a first line that spells the column names is consumed as a header. Positionally, the fields are the table's wire columns — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and every one of them must be present, in that order. An empty CSV field (and `\N` in TSV) takes the column's `DEFAULT`; too few fields is code 27, too many is 117, and under `header=absent` a header line is one record that fails to parse with code 27. Always the batch response shape. +- **Ingest accepts CSV and TSV bodies** (`internal/api/content_type.go`, `internal/api/ingest.go`, `internal/typelayer/{typelayer,ingest}.go`, `docs/src/content/docs/api.md`): `Content-Type: text/csv` and `text/tab-separated-values` are read by ClickHouse's own `CSV`/`TSV` readers. `text/csv; header=present` and `text/tab-separated-values; header=present` read `CSVWithNames` / `TSVWithNames` (the first line names the columns, in any order; an omitted column takes its `DEFAULT`; an unknown or duplicate name is code 117 for the whole request). `header` is RFC 4180 §3's optional parameter, mapped three ways onto ClickHouse: `header=present` is `CSVWithNames`, `header=absent` is strictly positional (`input_format_csv_detect_header=0` / `input_format_tsv_detect_header=0`), and a bare type is ClickHouse's default reading with header auto-detection on, so a first line that spells the column names is consumed as a header. Positionally, the fields are the table's wire columns — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column — and every one of them must be present, in that order. An empty CSV field (and `\N` in TSV, on a non-`Nullable` column — a `Nullable` one stores `NULL`) takes the column's `DEFAULT`; too few fields is code 27, too many is 117, and under `header=absent` a header line is read as a data row: refused with the code of the first column that cannot read its own name (27 for an `Int32`, 72 for a `Float64`), and accepted where every column can. Always the batch response shape. - **Schema discovery captures each table's DDL, its columns' ordinals and default expressions, and the server version** (`internal/discovery/discovery.go`, `internal/testutil/testutil.go`): `Column` gains `DefaultExpression` and `Position` (both from a widened `system.columns` select), `TableSchema` gains `DDL` from `system.tables.create_table_query`, and `SchemaRegistry` gains `ServerVersion()` from a `SELECT version()` probe next to the existing `SELECT timezone()`. Groundwork for the native type layer, captured on the same refresh as the columns so a stale version cannot outlive the schemas it describes. That is a publication guarantee, not a same-server one: `chconn.Manager` resolves the connection per call, so a reload changing `clickhouse.addr` mid-refresh can still pair a version from one server with schemas from another — narrow, and self-correcting on the next refresh. `DDL` is `json:"-"` and does **not** appear in `/v1/ops/schema`: that endpoint marshals `TableSchema` straight to the client, and an external-engine table (S3, MySQL, PostgreSQL, Kafka) renders its wiring there unconditionally — endpoint, bucket or host, database, username, S3 access key id. ClickHouse masks the password itself as `[HIDDEN]` from ~23.9 (verified on 26.7.3), so the exposure is the topology rather than the secret — except on an older server, or one with `display_secrets_in_show_and_select` enabled. `position` and `default_expression` are additive fields in the response. A table listed in `system.tables` with no `system.columns` rows is skipped rather than published column-less, and both new queries fail the refresh on error exactly as `timezone()` and `system.columns` do — callers keep the prior cache and retry. @@ -59,15 +59,15 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Logging goes through the `slog` default logger; no constructor takes a `*slog.Logger` anymore** (`internal/mq/embedded.go`, `internal/api/{ingest,pipes,structured_query,dlq,settings,errors,router}.go`, `internal/auth/auth.go`, `internal/discovery/{discovery,timestamp}.go`, `internal/ingest/{sweeper,worker}.go`, `internal/settings/{store,watch}.go`, `internal/chconn/chconn.go`, `internal/config/persistence.go`, `internal/app/wire.go`, `internal/testutil/logtest/` (new, + tests), `internal/testutil/testutil.go`): the general-notes refactor of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) and the cleanup deferred from [#586](https://github.com/Wave-RF/WaveHouse/pull/586), which left `internal/mq` logging half through an injected logger and half through the default. The logger parameter or field is gone from `mq.NewEmbedded`, `api.NewIngestHandler` / `NewPipesHandler` / `NewStructuredQueryHandler` / `NewDLQHandler` / `NewSettingsHandler`, `api.RequireAdmin` and `api.Dependencies.Logger`, `auth.NewAuthenticator`, `discovery.NewSchemaRegistry`, `ingest.NewSweeper` and the ingest worker, `settings.Open`, `chconn.Open` (whose field was never read), and `config.WarnIfFreshDataDir` / `LogStorageInitError`. Call sites use the context-aware calls (`slog.ErrorContext(ctx, …)`) wherever a context is in scope, so the trace handler can stamp them. Two visible differences: the ingest worker's lines no longer carry `component=ingest_worker`, and the auth middleware's operator-key audit lines and the settings reload lines are no longer skippable by passing a nil logger (only tests did). Tests reach log output through the new `internal/testutil/logtest`: `Silence()` from a package's `TestMain`, and `Capture(t, level)` for a test that asserts on log lines — which therefore runs serially, since the default logger is process-wide. `testutil.NopLogger` is removed. - **The process wiring moves out of `main.go` into `internal/app`** (`internal/app/` (new: `app.go`, `wire.go`, + tests), `cmd/wavehouse/main.go` (+ tests), `tests/integration/setup_test.go`, `.testcoverage.yml`, `.github/labeler.yml`): `app.New` builds every component from the boot config and the settings directory, `Run` drives the long-lived ones — ingest worker, sweeper, hub bridge, keepalive wheel, schema refresh, SIGHUP and the directory watcher, the API server and the Prometheus sidecar — under one `errgroup` until the signal context is cancelled or one of them fails, and `Close` releases what `New` opened in reverse order. Each component is wired in one place — what it opens, what it loops, what it releases — with the settings store handed to its wiring function whole, so the per-tenant registry ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)) lands there rather than in `main`. `main.go` shrinks to argv dispatch, the logger, `config.Load`, `CheckDataDir`, `app.New`, `app.Run`; `run(ctx)` takes the context `main` cancels on the first `SIGINT`/`SIGTERM`, so it is unit-tested end to end and the per-suite coverage exclude for it is gone. The integration suite boots through the same `app.New` against its testcontainer (the seed settings patched to the container, `default_role` set to the admin role) instead of a hand-built handler subset that had drifted from the binary. Behavior is unchanged except for the stop, which is now bounded end to end in three phases whose budgets add rather than multiply — `server.shutdown_timeout` for the drain, then fixed 5s and 3s for the release and the telemetry flush, so the worst case is the timeout plus 8s: `Run` drains the ingest worker and the API server's in-flight requests — the process could previously exit while the worker drain was still in flight — while open SSE streams are ended the moment the drain begins (a stream is a connection to close, not work to wait for; the client reconnects via `Last-Event-ID`) instead of holding the stop for the whole timeout; then `Close(ctx)` releases the stores under a context of its own, so a remote store's close can give up at the deadline rather than hang the exit, and finally flushes telemetry on a separate short budget so the flush that reports on the stop is never starved by a slow close. A settings reload caught mid-hook by the stop gives up with it. A second `SIGTERM`/`SIGINT` during the stop abandons it and exits non-zero, and `SIGHUP` is ignored once a stop has begun (it was briefly fatal: the reload loop's `signal.Stop` restored the default disposition at the start of the drain). The `mq.max_bytes_gb` reload hook bounds its JetStream calls to ten seconds, and when the DLQ resize fails it rolls the ingest stream back under a budget of its own instead of the one that just expired. `deployments/compose/standalone.yaml` sets `stop_grace_period` to cover all three phases, and the deployment docs gain a [Stopping](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md#stopping) section. Closes [#140](https://github.com/Wave-RF/WaveHouse/issues/140); story 0 of #583. -- **The chtypes SDK is `go/v0.5.2` (ABI revision 6), and the supported ClickHouse line is 26.8** (`go.mod`, `chtypes.lock`, `scripts/fetch-chtypes.sh`, `.github/actions/setup-env/action.yml`, `deployments/compose/*.yaml`, `tests/integration/setup_test.go`): the lock pins the revision-6 26.8.15.10 build (`b1790845279`) on darwin-arm64, linux-amd64 and linux-arm64, and must be regenerated whenever the SDK's ABI revision changes. The compose files, CI and the integration suite run `clickhouse/clickhouse-server:26.8.15.10`, off the retired 26.6 line. One 26.8 rule worth knowing: a bare JSON number in a `DateTime64` column is read as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — ClickHouse's rule, which WaveHouse stores as the server would. The default artifact cache moved to `~/.cache/chtypes/artifacts/abi6/-`, so the first fetch after upgrading downloads again (explicit `--dest` / `CHTYPES_REGISTRY` directories, as the Docker images use, are unaffected); CI's cache key and path carry the revision. Insert `check` clauses now cost one parse instead of two: they are judged inside the same `RowsExportWith` call that validates the body, through a compiled row filter, and the five ingest formats (JSON family, CSV, TSV, CSV and TSV with `header=present`) all take that path. +- **The chtypes SDK is `go/v0.5.2` (ABI revision 6), and the supported ClickHouse line is 26.8** (`go.mod`, `chtypes.lock`, `scripts/fetch-chtypes.sh`, `.github/actions/setup-env/action.yml`, `deployments/compose/*.yaml`, `tests/integration/setup_test.go`): the lock pins the revision-6 26.8.15.10 build (`b1790845279`) on darwin-arm64, linux-amd64 and linux-arm64, and must be regenerated whenever the SDK's ABI revision changes. The compose files, CI and the integration suite run `clickhouse/clickhouse-server:26.8.15.10`, off the retired 26.6 line. One 26.8 rule worth knowing: a bare JSON number in a `DateTime64` column is read as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — ClickHouse's rule, which WaveHouse stores as the server would. The default artifact cache moved to `~/.cache/chtypes/artifacts/abi6/-`, so the first fetch after upgrading downloads again (explicit `--dest` / `CHTYPES_REGISTRY` directories, as the Docker images use, are unaffected); CI's cache key and path carry the revision. Insert `check` clauses cost no second parse: they are judged inside the same `RowsExportWith` call that validates the body, through a compiled row filter, and the five ingest formats (JSON family, CSV, TSV, CSV and TSV with `header=present`) all take that path. -- **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `builds:` and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes --notes-start-tag "$(git describe --match 'v*')"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. +- **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `project_name`, `git.ignore_tags` and `builds:`, and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes` with `--notes-start-tag` set to the previous `v*` tag, `git describe --tags --abbrev=0 --match 'v*' "${tag}^"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. -- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser as-is, in one call per request, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable, `117` unknown field, `6` out of range — so `400 {"error":"invalid json"}` is gone from this endpoint; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — ingest responses, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date`, `Date32` and `Time` are unaffected) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A missing dedupe id is an absent column, a `null` cell or an empty string. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser in one call per request, as-is except that a JSON array is re-framed in place to one element per line so one bad element cannot lose the rest, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable (`33` for a record cut off mid-object), `117` unknown field, otherwise the type's own code (`72`, `41`, `38`, `69`, `376`, `467`, `691`, `675` …); an out-of-range integer wraps exactly as ClickHouse's own `INSERT` does (`256` → `0` in a `UInt8`) — so a malformed record is no longer the whole-request `400 {"error":"invalid json"}` (a JSON array whose brackets do not balance, or with content after its `]`, is still a whole-request `400 {"error":"invalid json: …"}`); a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — published rows, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date` and `Date32` keep their `"2026-06-21"` form on the stream, and on `/v1/query` and pipes they lose the `T00:00:00Z` the native path added — see the structured-query entry) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A dedupe id is read from the exported row: a `null` cell or an empty string is missing (so is an omitted `String` id with no `DEFAULT`), and a numeric id column cannot tell an omitted `0` from a supplied one. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. - **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, instead of walking a decoded record's keys. A column the role may not write stays in that schema as `MATERIALIZED` of its default, so naming it is refused while expressions that read it keep working and the stored row holds the table's default. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a non-`Nullable` checked column behaves exactly like omitting it (on a `Nullable` one it stores `NULL`, which fails the check). An `_eq` check also covers a column the role may not otherwise write: a record that omits it is filled with the required value and published, one that supplies exactly that value is accepted (**0.1.0 answered `403 column "x" not allowed for insert` to the correct value**), and any other value is `403 check failed for column "x"`. A role whose schema cannot be compiled this way, or that may write no column of the table, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After` (the cause is logged once a minute), rather than a `503` that a retry could not fix; a policy `check` on an `EPHEMERAL` column is still `403`. -- **WaveHouse now requires cgo, and supported platforms narrow to darwin/arm64, linux/amd64, linux/arm64** (BREAKING; `go.mod`, `scripts/build.sh`, `.goreleaser.yaml`, `deployments/Dockerfile`, `deployments/Dockerfile.goreleaser`, `Makefile`, `internal/config/config.go`, `config.yaml`, `chtypes.lock` (new), `scripts/fetch-chtypes.sh` (new), `.github/actions/setup-env/action.yml`, `.github/workflows/ci.yml`): the native type layer above needs cgo for `dlfcn` (no C library linked, no header). cgo is now unconditional: `CGO_ENABLED=0` is gone from every build path, and the `make audit-cgo` target that policed the old no-cgo build has been **removed** along with it (`make binary-analysis` is now `size` + `deadcode`). Because chtypes publishes artifacts only for darwin-arm64, linux-amd64 and linux-arm64, **Windows, FreeBSD and darwin/amd64 builds are discontinued** — `.goreleaser.yaml`'s matrix drops from 8 targets to 3, and the release archives/checksums/GHCR image narrow to match. The runtime image moves from an Alpine/musl builder + `distroless/static` to `golang:1.27-bookworm` (glibc, ships gcc) + `distroless/cc-debian12` (glibc + libstdc++, which the SDK's shared library needs) and bakes the pinned chtypes artifact into the image at `/opt/chtypes/artifacts` via a new `chtypes.lock` (exact file + sha256 per platform/line) and `scripts/fetch-chtypes.sh --frozen` wrapper, so the container has no first-request download. `go.mod` moves to `go 1.27`. Only processes running the `api` role load the artifact, so an ingest-worker-only or sweeper-only process needs none installed; the glibc requirement is 2.34 or later. New boot config: `clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY` lets an operator point at an explicit registry directory instead of the SDK's own search path (the shipped image instead sets the SDK's own `CHTYPES_REGISTRY` env var directly). CI's `unit`/`integration`/`e2e` jobs fetch and cache the pinned artifact (`setup-env`'s new `chtypes` input) and set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` so a missing artifact fails the job instead of silently skipping the chtypes-backed tests. `GOLANGCI_LINT_VERSION` bumped `v2.11.4` → `v2.13.2`: the `go 1.27` bump panics `v2.11.4`'s type checker on every package; `v2.13.0` is the oldest release whose changelog claims go1.27 support, but it panics in this tree for a different reason (`nilness`/`honnef.co/go/tools@v0.8.0-rc.1` crashing while analyzing a third-party dependency), fixed once that dependency moves past its release candidate in `v2.13.1`. The cross-toolchain approach this bullet originally described was replaced before landing — see the release-pipeline entry above. +- **WaveHouse now requires cgo, and supported platforms narrow to darwin/arm64, linux/amd64, linux/arm64** (BREAKING; `go.mod`, `scripts/build.sh`, `.goreleaser.yaml`, `deployments/Dockerfile`, `deployments/Dockerfile.goreleaser`, `Makefile`, `internal/config/config.go`, `config.yaml`, `chtypes.lock` (new), `scripts/fetch-chtypes.sh` (new), `.github/actions/setup-env/action.yml`, `.github/workflows/ci.yml`): the native type layer above needs cgo for `dlfcn` (no C library linked, no header). cgo is now unconditional: `CGO_ENABLED=0` is gone from every build path, and the `make audit-cgo` target that policed the old no-cgo build has been **removed** along with it (`make binary-analysis` is now `size` + `deadcode`). Because chtypes publishes artifacts only for darwin-arm64, linux-amd64 and linux-arm64, **Windows, FreeBSD and darwin/amd64 builds are discontinued** — `.goreleaser.yaml`'s matrix drops from 8 targets to 3, and the release archives/checksums/GHCR image narrow to match. The runtime image moves from an Alpine/musl builder + `distroless/static` to `golang:1.27-bookworm` (glibc, ships gcc) + `distroless/cc-debian12` (glibc + libstdc++, which the SDK's shared library needs) and bakes the pinned chtypes artifact into the image at `/opt/chtypes/artifacts` via a new `chtypes.lock` (exact file + sha256 per platform/line) and the `scripts/fetch-chtypes.sh` wrapper (which always fetches with `--frozen --lock chtypes.lock`), so the container has no first-request download. `go.mod` moves to `go 1.27`. Only processes running the `api` role load the artifact, so an ingest-worker-only or sweeper-only process needs none installed; the glibc requirement is 2.34 or later. New boot config: `clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY` lets an operator point at an explicit registry directory instead of the SDK's own search path (the shipped image instead sets the SDK's own `CHTYPES_REGISTRY` env var directly). CI's `unit`/`integration`/`e2e` jobs fetch and cache the pinned artifact (`setup-env`'s new `chtypes` input) and set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` so a missing artifact fails the job instead of silently skipping the chtypes-backed tests. `GOLANGCI_LINT_VERSION` bumped `v2.11.4` → `v2.13.2`: the `go 1.27` bump panics `v2.11.4`'s type checker on every package; `v2.13.0` is the oldest release whose changelog claims go1.27 support, but it panics in this tree for a different reason (`nilness`/`honnef.co/go/tools@v0.8.0-rc.1` crashing while analyzing a third-party dependency), fixed once that dependency moves past its release candidate in `v2.13.1`. The cross-toolchain approach this bullet originally described was replaced before landing — see the release-pipeline entry above. - **Boot refuses an unbound `WH_*` environment variable and an unusable `data_dir`** (BREAKING; `internal/config/check.go` (new, + tests), `internal/config/{config,persistence}.go`, `cmd/wavehouse/main.go`, `docs/src/integrations/diagram-png.mjs`): the environment half of the strict YAML loader. `config.Load` now errors, naming every offender, on a `WH_*` variable that no `Config` field binds — the two variables read outside the struct, `WH_CONFIG` and `WH_LOG_LEVEL`, are exempt — `WH_DEDUPE_ENABLED=true` left in a compose file from before the settings-directory move, or a misspelling, was set, ignored, and believed. **An existing deployment that still exports a variable this release moved to the settings directory stops booting until it is unset**; the upgrade runbook in `deployment.md` gains that audit. Only the `WH_` prefix is checked, since the environment always carries unrelated names; the one outside source that shares it — Kubernetes service-link variables for a Service named `wh` or `wh-*` — is named in the error with the `enableServiceLinks: false` remediation, and the docs build's opt-out knob is renamed from `WH_SKIP_DIAGRAM_PNG` to `DOCS_SKIP_DIAGRAM_PNG` so an exported one no longer refuses a local boot. Right after `Load`, before ClickHouse or the settings directory are touched, `config.CheckDataDir` probes `data_dir` and refuses boot on any of: an empty or blank value (reachable through `WH_DATA_DIR=`), refused outright since the ancestor walk would otherwise fall back to the working directory and NATS and Pebble state would land under it; a path that exists and is not a directory; a dangling symlink at `data_dir` or any component above it (the walk to the nearest existing ancestor uses `Lstat`, so a failed mount is not skipped over as "does not exist" and passed in an unrelated directory); and a directory the process cannot write to — or, when it does not exist, an unwritable nearest ancestor — probed by creating and removing one temp file. So an unusable `data_dir` refuses boot before schema discovery rather than after it; a permission denial — on the probe, or on reaching the path at all through a parent without search permission — carries the UID-65532 remediation (a bind mount owned by root is the typical cause), and that hint string is now shared with `LogStorageInitError`. `EnvConfig` and `EnvLogLevel` join `EnvSettingsDir` as the exported names for the process-level variables. Boot is the validator for the non-hot-reloadable half — there is no dry-run subcommand, by decision on #530: boot config only takes effect through a restart, so the restart is where it is checked, and the docs say so. Closes #530. @@ -77,18 +77,18 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Policy format v2: `policies.json` is role-first, and the two operations are separate permission types** (BREAKING; `internal/policy/policy.go`, `internal/policy/rowfilter.go`, `internal/settings/validate.go`, `internal/query/builder.go`, `internal/api/{ingest,structured_query}.go`, `internal/stream/hub.go`, `clients/ts/src/{types,index}.ts`, `deployments/compose/settings/policies.json`, `docs/src/content/docs/{access-control.mdx,settings-directory.mdx,architecture.md}`, `AGENTS.md`, `tests/e2e/sdk/`): a table entry was keyed `tables.
.select.`; it is now keyed `tables.
..select`. A role appears once per table and its grant carries two optional blocks, so a role that could both read and write no longer has to be written out twice, and "this role has no insert grant" is a missing block rather than an absence you have to notice in a second map. The blocks are now distinct types rather than one struct whose halves were inert per operation: `select` takes `allow_columns`, `deny_columns`, `filter`, `allowed_aggregations`, `denied_aggregations` and the four `max_*` limits; `insert` takes `allow_columns`, `deny_columns`, `check`. Field names and semantics are unchanged — only the nesting moves — but a field on the wrong side that the old layout *accepted* — the four `max_*` limits and the two aggregation rules — is now a validation error instead of being silently ignored. `filter` under an `insert` grant and `check` under a `select` one are rejected too, but that is not new here — [#541](https://github.com/Wave-RF/WaveHouse/pull/541), also unreleased, added the runtime check; the split types now refuse them one layer earlier, as unknown keys at the strict decode. Upgrading from **0.1.0**, though, none of the eight were enforced: a 0.1.0 policy could carry an insert-side `filter` that resolved into a `WHERE` the insert path never read, as well as an ignored limit. Converting to the role-first layout drops both. Internally `ResolvedPermissions` splits the same way (`.Select` / `.Insert`), and `IsColumnAllowed` takes the side to consult, which is what stops the read allowlist from ever answering a write question or vice versa. **There is no automatic conversion** — the settings files are the source of truth and WaveHouse has no write path back to them — so `policies.json` must be converted by hand; run `wavehouse validate` before restarting. A document still in the old layout is reported as one clear finding naming the table and operation and pointing at [the migration note](https://wavehouse.dev/access-control#migrating-from-the-operation-first-layout), instead of the confusing strict-decode "unknown field" error it would otherwise produce (or, for an empty operation block, silently decoding as a role named `select` with no grants — which fails the undeclared-role check when `roles.json` does not declare a role named `select` — the usual case, since `checkRoleRefs` errors and moves on before reaching the "grant sets neither select nor insert" warning. If such a role *is* declared, you get that warning instead and the document adopts). One shape is refused differently: a grant keyed by a role named after the *other* operation (`tables.t.select.insert`) reads as a different grant under each layout — different role, different operation, or both — so it gets its own error asking you to rename the role rather than the migration pointer. A role named after its *own* operation (`tables.t.select.select`) means the same thing either way and is accepted. -- **Ingest now requires a declared `Content-Type`, and it is authoritative** (BREAKING; `internal/api/record_reader.go`, `internal/api/ingest.go`, `clients/ts/src/table.ts`, `docs/src/content/docs/{api.md,architecture.md,sdk/queries.md}`): `POST /v1/ingest` used to sniff the body and treat the header as a hint — the first non-whitespace byte chose between a single object and an array, and an `application/x-ndjson` body that happened to start with `[` was silently re-read as a JSON array. A request that declares **no** `Content-Type`, or one whose media type is not in the accepted list, is now rejected with `415` before the body is parsed, naming every accepted type (`application/json`, `application/x-ndjson`, `application/ndjson`, `application/jsonl`, `application/jsonlines`) and quoting what was declared, bounded to four distinct header lines each capped at 128 bytes. The header is parsed per RFC 9110 §8.3 via `mime.ParseMediaType` rather than by hand, so the grammar's rules apply — parameters never affect the format (with the one exception below), and a comma inside a quoted value is data. Because `Content-Type` is a **singleton** field (§5.3 forbids repeating it), anything that is not exactly one readable media type is refused; the one accommodation is that repeated header lines are all resolved and accepted when they agree, since honoring just the first would let an NDJSON body be read as one JSON object and drop every record past it. A comma-joined value gets no such accommodation — §8.3 warns that taking a member of the pseudo-list is itself an interoperability and security hazard. **Four additional shapes 0.1.0 accepted now `415`** (beyond repeated header lines that disagree, which it also accepted): a present-but-empty header (a bare `Content-Type:` line, or one that is only whitespace); a value with a trailing or leading comma (`application/json,`); a comma-joined value that does not parse as a single media type (`application/json, application/json` — but a comma *inside a quoted parameter value* is legal data, so `application/json; a=", application/x-ndjson; b="` is one media type and is accepted); and a malformed parameter on a line that *also* carries a comma, which is refused rather than guessed at because the comma may be a second declaration joined on — so `application/json; profile="a,b"; charset` is a `415` while `application/json; profile="a,b"` and `application/json; charset` are each accepted ([#563](https://github.com/Wave-RF/WaveHouse/issues/563)). A repeated parameter name is *not* among them: the media type is re-parsed alone, so `; charset=a; charset=b` reads as `application/json` like every other malformed parameter. The declaration now decides the format outright: a body declared as NDJSON is read as NDJSON whatever its first byte, so a line that isn't a JSON object fails as a **per-record** error through the existing batch-result path instead of re-framing the whole request. The body still picks arity *within* the JSON family — `[` is an array, anything else a single object — because those are the same format at different lengths. Clients that relied on the sniffer must now send a header; the TS SDK already sent one on both paths (`application/json` for a single object, `application/x-ndjson` for arrays and `insertNDJSON`) and now states it at each call site rather than leaning on the request default, and every `curl` example in the docs already carried one. The format is modeled as an `IngestFormat` where the sniffing used to live, with the slot for CSV kept where the old comment marked it. +- **Ingest now requires a declared `Content-Type`, and it is authoritative** (BREAKING; `internal/api/record_reader.go`, `internal/api/ingest.go`, `clients/ts/src/table.ts`, `docs/src/content/docs/{api.md,architecture.md,sdk/queries.md}`): `POST /v1/ingest` used to sniff the body and treat the header as a hint — the first non-whitespace byte chose between a single object and an array, and an `application/x-ndjson` body that happened to start with `[` was silently re-read as a JSON array. A request that declares **no** `Content-Type`, or one whose media type is not in the accepted list, is now rejected with `415` before the body is parsed, naming every accepted type (`application/json`, `application/x-ndjson`, `application/ndjson`, `application/jsonl`, `application/jsonlines`) and quoting what was declared, bounded to four distinct header lines each capped at 128 bytes. The header is parsed per RFC 9110 §8.3 via `mime.ParseMediaType` rather than by hand, so the grammar's rules apply — parameters never affect the format (with the one exception below), and a comma inside a quoted value is data. Because `Content-Type` is a **singleton** field (§5.3 forbids repeating it), anything that is not exactly one readable media type is refused; the one accommodation is that repeated header lines are all resolved and accepted when they agree, since honoring just the first would let an NDJSON body be read as one JSON object and drop every record past it. A comma-joined value gets no such accommodation — §8.3 warns that taking a member of the pseudo-list is itself an interoperability and security hazard. **Four additional shapes 0.1.0 accepted now `415`** (beyond repeated header lines that disagree, which it also accepted): a present-but-empty header (a bare `Content-Type:` line, or one that is only whitespace); a value with a trailing or leading comma (`application/json,`); a comma-joined value that does not parse as a single media type (`application/json, application/json` — but a comma *inside a quoted parameter value* is legal data, so `application/json; a=", application/x-ndjson; b="` is one media type and is accepted); and a malformed parameter on a line that *also* carries a comma, which is refused rather than guessed at because the comma may be a second declaration joined on — so `application/json; profile="a,b"; charset` is a `415` while `application/json; profile="a,b"` and `application/json; charset` are each accepted ([#563](https://github.com/Wave-RF/WaveHouse/issues/563)). A repeated parameter name is *not* among them: the media type is re-parsed alone, so `; charset=a; charset=b` reads as `application/json` like every other malformed parameter. The declaration now decides the format outright: a body declared as NDJSON is read as NDJSON whatever its first byte, so a line that isn't a JSON object fails as a **per-record** error through the existing batch-result path instead of re-framing the whole request. The body still picks arity *within* the JSON family — `[` is an array, anything else a single object — because those are the same format at different lengths. Clients that relied on the sniffer must now send a header; the TS SDK already sent one on both paths (`application/json` for a single object, `application/x-ndjson` for arrays and `insertNDJSON`) and now states it at each call site rather than leaning on the request default, and every `curl` example in the docs already carried one. The format is modeled as an `IngestFormat` where the sniffing used to live, with the slot for CSV kept where the old comment marked it. (Superseded in part: `record_reader.go` is now `internal/api/content_type.go`, the accepted list also carries `text/csv` and `text/tab-separated-values`, each bare or with `header=present`/`header=absent`, and on those two types the `header` parameter selects the format — see the CSV/TSV entry.) - **An `EPHEMERAL` column is accepted as ingest input only where it can be honoured** (`internal/typelayer/{typelayer,ingest}.go`, `internal/api/ingest.go`, `docs/src/content/docs/{api.md,sdk/reference.md,deployment.md,architecture.md}`): a record in a JSON body, or in CSV/TSV with `header=present`, may name an `EPHEMERAL` column when the role may write it, a `DEFAULT` column reads it and no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` expression does; its value feeds that `DEFAULT` and is never stored, selected or published. Any other `EPHEMERAL` column is refused like an unknown one (`400`, code 117), because ClickHouse computes `MATERIALIZED` columns from the published row, which never carries the ephemeral value. Positional CSV/TSV (no header, or `header=absent`) carry the wire columns only. A policy `check` on an `EPHEMERAL` column is still `403`. - **A JSON array body with content after its closing `]` is refused** (`internal/api/ingest_framing.go`): the request answers `400 {"error":"invalid json: content after the closing ']' of the json array"}` and publishes nothing. -- **A role's ingest handle pool grows under load** (`internal/typelayer/pool.go`): a role shape's pool of compiled handles starts at one and grows to `min(GOMAXPROCS, 4)` when every handle is busy, as a base table's grows to `min(GOMAXPROCS, 8)`, so concurrent inserts by one role no longer serialize on a single handle. +- **A role's ingest handle pool grows under load** (`internal/typelayer/pool.go`): a role shape's pool of compiled handles starts at one and grows to `min(GOMAXPROCS, 4)` when every handle is busy, as a base table's grows to `min(GOMAXPROCS, 8)`, so concurrent inserts by one role do not serialize on a single handle. - **Client-side timestamp comparison in the SDK is by instant** (`clients/ts/src/timestamp.ts` (new), `clients/ts/src/{query-builder,stream/live-query}.ts`): two strings that both read as timestamps are compared to the nanosecond for `=`, `!=`, `in`, `>`, `>=`, `<`, `<=` in stream `where` filters, with a zone-less value read as UTC; the live-query backfill seam compares at the coarser of the two precisions, so an event in the boundary row's own millisecond is covered. Other strings keep strict equality and string order. -- **Test helpers moved to `internal/typelayer/typelayertest`** (internal; `.testcoverage.yml`, `.github/labeler.yml`): `TestEngine`, `SkipWithoutArtifact` and `RequireEnv` no longer live in `internal/typelayer`, which now never imports `testing`. The helper package is excluded from coverage like the other test-only packages, and a PR touching `internal/typelayer/`, `chtypes.lock` or `scripts/fetch-chtypes.sh` is labeled `area/typelayer`. -- **A policy `check` on a column the table cannot accept is now refused instead of silently unenforced** (BREAKING; `internal/api/ingest.go`, `internal/discovery/discovery.go`, `docs/src/content/docs/{api.md,access-control.mdx}`): a `check` clause naming a column the table does not have, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one can never be enforced — the published row carries one slot per insertable column, so an auto-injected value for anything outside that set is dropped on the way out, and an ephemeral column is never stored even though the row does carry it. The record inserted **without** the value the policy required and answered `200 {"ok":true}`. Demonstrated on this branch: a `check` of `tenant _eq {{ jwt.tenant }}` against a `MATERIALIZED tenant` column published `columns:["page"], row:["/a"]` — the tenant constraint absent from the row entirely. It is now refused, naming every offending column and the reason, on **every** insert by that role until the policy or the table is corrected: a single-object request answers `403`, while a batch answers `200` with the same message against each record in `results` — the batch is still read to the end and reports per record, as it does for any other rejection. Policy validation cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**; `wavehouse validate` will not tell you. Superseded in part: on this branch the published row never carries an ephemeral value (see the `EPHEMERAL` entry above). +- **Test helpers live in `internal/typelayer/typelayertest`** (internal; `.testcoverage.yml`, `.github/labeler.yml`): `TestEngine`, `SkipWithoutArtifact` and `RequireEnv` sit outside `internal/typelayer`, which never imports `testing`. The helper package is excluded from coverage like the other test-only packages, and a PR touching `internal/typelayer/`, `chtypes.lock` or `scripts/fetch-chtypes.sh` is labeled `area/typelayer`. +- **A policy `check` on a column the table cannot accept is now refused instead of silently unenforced** (BREAKING; `internal/api/ingest.go`, `internal/discovery/discovery.go`, `docs/src/content/docs/{api.md,access-control.mdx}`): a `check` clause naming a column the table does not have, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one can never be enforced — the published row carries one slot per insertable column, so an auto-injected value for anything outside that set is dropped on the way out, and an ephemeral column is never stored even though the row does carry it. The record inserted **without** the value the policy required and answered `200 {"ok":true}`. Demonstrated on this branch: a `check` of `tenant _eq {{ jwt.tenant }}` against a `MATERIALIZED tenant` column published `columns:["page"], row:["/a"]` — the tenant constraint absent from the row entirely. It is now refused, naming every offending column and the reason, on **every** insert by that role until the policy or the table is corrected: a single-object request answers `403`, while a batch answers `200` with the same message against each record in `results` — the batch is still read to the end and reports per record, as it does for any other rejection. Policy validation cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**; `wavehouse validate` will not tell you. Superseded in part: in this release the published row never carries an ephemeral value (see the `EPHEMERAL` entry above). -- **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind`; `TableSchema` gains `IsInsertable` (the `InsertableColumns` / `InsertableColumnNames` lists it first added are removed again, an internal API: which columns a record may supply is the type layer's `Table.WireColumns`). `EPHEMERAL` columns are likewise left out of the envelope, and this entry's earlier claim that they stay insertable is superseded by the `EPHEMERAL` entry above. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. Superseded: that rejection is now ClickHouse's own `400` with `exception_code` 117, `Unknown field found while parsing …` (see the type-layer entry). +- **The ingest envelope carries only insertable columns** (BREAKING; `internal/discovery/discovery.go`, `internal/api/ingest.go`, `internal/stream/hub.go`, `internal/testutil/testutil.go`, `tests/integration/ingest_test.go`, `clients/ts/src/types.ts`): naming columns explicitly in the `INSERT` — the change above — makes a computed column fatal, so the envelope, the compact encoder and the SSE connect-time announcement now use the table's **insertable** subset. Verified against ClickHouse 26.6.3: a `MATERIALIZED` column in an `INSERT` column list is `Cannot insert column …, because it is MATERIALIZED column` (code 44, and `insert_allow_materialized_columns` defaults to `0`); an `ALIAS` column is `No such column …` (code 16). Schema discovery reads every row of `system.columns` with no `default_kind` filter, so without this both would land in the envelope and then in the statement, and **a table carrying either could ingest under the previous column-less `FORMAT JSONEachRow` and could not ingest at all** — every row to the DLQ, or redelivered forever where the DLQ is off. `Column` gains `DefaultKind` and `IsInsertable` (the `InsertableColumns` / `InsertableColumnNames` lists it first added are removed again, an internal API: which columns the envelope carries is the type layer's `Table.WireColumns`). `EPHEMERAL` columns are likewise left out of the envelope, and this entry's earlier claim that they stay insertable is superseded by the `EPHEMERAL` entry above. `GET /v1/ops/schema` still reports the whole table, now including `default_kind`, `default_expression` and `position`: a computed column stays queryable, it just cannot be written. No fixture in the suite declared a computed column, which is why every gate was green while this was broken; `tests/integration` now creates one and drives HTTP ingest → NATS → the worker's `INSERT` end to end. **BREAKING:** a record that *supplies* a value for a `MATERIALIZED`/`ALIAS` column is now rejected (`400 … cannot be inserted`) where it was previously accepted and silently dropped by the positional encoder. Superseded: that rejection is now ClickHouse's own `400` with `exception_code` 117, `Unknown field found while parsing …` (see the type-layer entry). -- **Ingest reads the request body up front, and the per-record decisions sit behind interfaces** (`internal/api/{ingest,ingest_seams,bufpool,record_reader}.go`, `internal/stream/hub.go`, `internal/ingest/compact.go`): responses are unchanged except at the body cap and one new read-failure body (`400 {"error":"invalid request body"}`, when the body cannot be read at all — a malformed transfer encoding or a truncated upload, which previously surfaced through the decoder as `invalid json`), and at the cap the `413` is now decided before any record is processed: an over-cap batch no longer ingests the prefix it had already decoded, and an over-cap single-object body whose first object was followed by an oversized tail — which used to answer `200` after ingesting that one object — now answers `413`. Both are improvements, since a client retrying a `413` can no longer double-insert a prefix, but they are behavior changes and the memory profile changes too (see below); this is the seam work the native type layer lands against. The handler now reads the whole (already `MaxBytesReader`-capped) body into a pooled `*bytes.Buffer` and runs the record readers over those bytes rather than the live connection — so the `413` surfaces at that read instead of mid-iteration (same status, same message), and the `415` is decided from the header before a single byte is read. Three decision points became interfaces with default implementations that delegate to today's code unchanged: `RecordValidator` (schema validation + timestamp canonicalization — the two calls stay where they are, with the check-clause block between them, since merging them would move checks onto canonicalized values), `InsertChecker` (the `_eq` and `_in` comparisons), and `stream.RowEvaluator` (row visibility, reached by both the live fan-out and replay through the one shared admission step). **The memory profile is not unchanged, and that is the deliberate part.** Streaming meant peak resident bytes on the order of one record: the NDJSON path scanned line by line and the array path let `json.Decoder` compact after each element. Peak is now O(body) per in-flight request — and `bytes.Buffer` grows by doubling, so the peak allocation can exceed the body cap before `MaxBytesReader` errors. `maxPooledBufferBytes` (1 MiB) caps what a request hands *back* to the pool, not its peak, and nothing in `internal/api` bounds total in-flight bytes, so the ceiling is concurrency × the 16 MiB data-plane cap — which has no operator knob (`maxRequestBytes` is test-only), so the outer limit is the reverse proxy's, which the reverse-proxy guide already advises setting. Kept because it is the shape the native type layer lands against, which needs the body addressable rather than consumed; a bound on total in-flight ingest bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). Operators fronting large batches at high concurrency should size for it or cap body size at the proxy. All three are nil-safe: an un-wired handler or `Hub` uses the default rather than panicking past the check. Also new: `ingest.EncodeCompactRow`, which renders a record as one `JSONCompactEachRow` line — inert in this commit, and the encoder every published row went through at that point in the release (superseded: the published row is now ClickHouse's own export, and `EncodeCompactRow`, `RecordValidator` and `InsertChecker` are gone — see the type-layer entry). +- **Ingest reads the request body up front, and the per-record decisions sit behind interfaces** (`internal/api/{ingest,ingest_seams,bufpool,record_reader}.go`, `internal/stream/hub.go`, `internal/ingest/compact.go`): responses are unchanged except at the body cap and one new read-failure body (`400 {"error":"invalid request body"}`, when the body cannot be read at all — a malformed transfer encoding or a truncated upload, which previously surfaced through the decoder as `invalid json`), and at the cap the `413` is now decided before any record is processed: an over-cap batch no longer ingests the prefix it had already decoded, and an over-cap single-object body whose first object was followed by an oversized tail — which used to answer `200` after ingesting that one object — now answers `413`. Both are improvements, since a client retrying a `413` can no longer double-insert a prefix, but they are behavior changes and the memory profile changes too (see below); this is the seam work the native type layer lands against. The handler now reads the whole (already `MaxBytesReader`-capped) body into a pooled `*bytes.Buffer` and runs the record readers over those bytes rather than the live connection — so the `413` surfaces at that read instead of mid-iteration (same status, same message), and the `415` is decided from the header before a single byte is read. Three decision points became interfaces with default implementations that delegate to today's code unchanged: `RecordValidator` (schema validation + timestamp canonicalization — the two calls stay where they are, with the check-clause block between them, since merging them would move checks onto canonicalized values), `InsertChecker` (the `_eq` and `_in` comparisons), and `stream.RowEvaluator` (row visibility, reached by both the live fan-out and replay through the one shared admission step). **The memory profile is not unchanged, and that is the deliberate part.** Streaming meant peak resident bytes on the order of one record: the NDJSON path scanned line by line and the array path let `json.Decoder` compact after each element. Peak is now O(body) per in-flight request — and `bytes.Buffer` grows by doubling, so the peak allocation can exceed the body cap before `MaxBytesReader` errors. `maxPooledBufferBytes` (1 MiB) caps what a request hands *back* to the pool, not its peak, and nothing in `internal/api` bounds total in-flight bytes, so the ceiling is concurrency × the 16 MiB data-plane cap — which has no operator knob (`maxRequestBytes` is test-only), so the outer limit is the reverse proxy's, which the reverse-proxy guide already advises setting. Kept because it is the shape the native type layer lands against, which needs the body addressable rather than consumed; a bound on total in-flight ingest bytes is tracked in [#544](https://github.com/Wave-RF/WaveHouse/issues/544). Operators fronting large batches at high concurrency should size for it or cap body size at the proxy. All three are nil-safe: an un-wired handler or `Hub` uses the default rather than panicking past the check. Also new: `ingest.EncodeCompactRow`, which renders a record as one `JSONCompactEachRow` line — inert in this commit, and the encoder every published row went through at that point in the release (superseded: the published row is now ClickHouse's own export, and `EncodeCompactRow`, `RecordValidator`, `InsertChecker` and the record readers are gone — see the type-layer entry). - **NATS envelope v2: the row travels positionally, with the column names sent alongside** (BREAKING; `internal/ingest/{types,compact,worker}.go`, `internal/api/ingest.go`, `docs/src/content/docs/{api.md,architecture.md,ingest-pipeline.md}`): `EventMessage`'s `data` object is replaced by `format` (`"JSONCompactEachRow"`), `columns` (the table's declaration order) and `row` (one compact line — a positional JSON array). The `INSERT` the worker emits carries the column names once for a whole group instead of every row repeating every key (each NATS envelope still carries its own `columns`, since a message must stand alone), and a reader can tell a schema change mid-stream from a reordering. **In-flight NATS messages published by an older version are not readable by the new worker** — an envelope whose `format` is absent or unknown, or whose `columns` and `row` can't be paired (a length mismatch, an undecodable row), carries no way to say which value belongs to which column. Such an envelope is parked on the DLQ (see the Fixed entry below), never inserted. **Drain the ingest queue before deploying.** The worker groups a batch by column list, so a schema change mid-stream splits the INSERT rather than corrupting it, and writes `INSERT INTO {table} (cols) FORMAT JSONCompactEachRow` — the table still binds as a server-side `Identifier` parameter, while the column list, which has no such parameter, is quoted client-side by the same `chsql.QuoteIdent` the query builder uses. A positional row has one value per column and no way to say "absent", so a field the record omitted now rides as an explicit `null` in its slot; `input_format_null_as_default=1` (already the server default — set explicitly for one configured otherwise) turns that back into the column's default for a **non-nullable** column, matching what omitting the key did under `JSONEachRow`. **Transitional divergence, on `Nullable` columns only, and it is not what that setting controls:** ClickHouse stores an explicit `null` as `NULL` on a nullable column whatever the setting says — only an *absent* key ever took the default — so a `Nullable(T) DEFAULT …` column now stores `NULL` where it previously took its default. Verified against ClickHouse 26.6.3 (omitted key → default; explicit null → `NULL` at either setting). *Against a server explicitly running `input_format_null_as_default=0`*, the reverse also changes: an explicit `null` for a non-nullable column with a default now takes the default rather than failing the row into the DLQ, because WaveHouse pins the setting instead of inheriting it. On a default-configured server that was already the behavior. Every other column type behaves as before. The DLQ flow is unchanged — its payload is the new envelope. Row cells are copied as their original bytes rather than re-encoded at each hop, so a 64-bit id past 2^53 keeps every digit end to end. (Superseded for the omitted-field behaviour: the row is now ClickHouse's own export with `DEFAULT`s already evaluated, so a `Nullable(T) DEFAULT …` column takes its default again and the divergence above no longer exists — see the type-layer entry.) @@ -98,13 +98,13 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The landing page's live demo now reads from the stats deployment's new WaveHouse Cloud backend** (`docs/src/components/LiveDemo.astro`, `docs/scripts/screenshot.mjs`): the GitHub-activity dogfood deployment behind the hero panel (Wave-RF/WaveHouse-Stats) moved off its self-managed AWS infrastructure onto WaveHouse Cloud, so `BASE_URL` — the origin `@wavehouse/sdk` queries in the visitor's browser — points at `https://iefrrvavd5akvphk7pq3.wavehouse.app` instead of `https://stats.wavehouse.dev`, ahead of the AWS stack being torn down. The `PUBLIC_WAVEHOUSE_STATS_URL` build-time override is unchanged, so a fork or staging docs build still redirects the panel without a code edit. **`DEMO_HOST` deliberately stays `stats.wavehouse.dev`** — the demo *site* is still served there and is still what the panel's chrome label and "Full demo" link should show; the migration splits the site from the API origin behind it, and the two constants now carry comments saying so. Verified against the new deployment before the switch: all five pipes the panel reads (`gh_summary`, `gh_activity_recent`, `gh_events_per_minute`, and the pre-#19 `gh_stars_total` / `gh_forks_total` fallbacks) return `200` with the same row shapes, the structured-query backfill fallback (`POST /v1/query?table=gh_events`) matches its old-backend response byte for byte, `GET /v1/stream?table=gh_events` opens an SSE stream, and CORS is unchanged (`Access-Control-Allow-Origin: *`, `X-Cache` exposed) so the cross-origin browser reads keep working from the docs site. The new backend is already the live ingest target — it reported more recent events than the old one at cutover (5,175 vs 5,038 over 7d) — which is the other half of why the panel had to follow it. `screenshot.mjs`'s `networkidle` note is retargeted to "the stats demo backend" rather than naming a host it no longer connects to. -- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/clickhouse_exec.go` (trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. The reader's HTTP connections are capped per connection tuple (URL, user, password, database, TLS) at the largest `max_open_conns` among the tenants sharing it, and its requests carry the tenant's `clickhouse.headers`; the settings warning about a plaintext HTTP hop (`internal/settings/validate.go`) now says it carries the credentials on every query and insert. The 64 MiB cap is new on these two paths, so a pipe (which has no row limit) returning more than 64 MiB now fails; `/v1/query` normally stays under it through its default row cap. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. +- **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/sql_classify.go` (was `clickhouse_exec.go`, trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`, `internal/settings/validate.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), `Date`/`Date32` are `"2026-06-21"` where the native driver gave `"2026-06-21T00:00:00Z"`, and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. The reader's HTTP connections are capped per connection tuple (URL, user, password, database, TLS) at the largest `max_open_conns` among the tenants sharing it, and its requests carry the tenant's `clickhouse.headers`; the settings warning about a plaintext HTTP hop (`internal/settings/validate.go`) now says it carries the credentials on every query and insert. The 64 MiB cap is new on these two paths, so a pipe (which has no row limit) returning more than 64 MiB now fails; `/v1/query` normally stays under it through its default row cap. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. - **WH001 (no hard-wrapped prose) now applies to every tracked Markdown file, with no carve-out** (`.github/.markdownlint.json` (deleted), `.claude/.markdownlint.json` (deleted), `.claude/skills/integration-astro-view-transitions/` (deleted), `.markdownlint-cli2.jsonc`, `.github/workflows/README.md`, `.claude/skills/pm-triage/references/routine.md`, `AGENTS.md`, `scripts/docs-prose.sh`, `.github/prompts/docs-review.md`, `docs/src/content/docs/claude-code.md`, `docs/src/content/docs/development.md`, `.claude/agents/docs-reviewer.md`): two path-scoped configs had switched WH001 off under `.github/` and `.claude/` ever since [#489](https://github.com/Wave-RF/WaveHouse/pull/489) introduced the rule — baked in from the start rather than added in response to a discovered problem — which left the repo documenting the rule three ways and disagreeing with itself: `CONTRIBUTING.md` promises contributors `make lint` enforces it *everywhere*, while `AGENTS.md` and the `.markdownlint-cli2.jsonc` header wrote up the carve-out. Not theoretical: on [#520](https://github.com/Wave-RF/WaveHouse/pull/520) a reviewer correctly flagged a hard-wrapped bullet in `.github/workflows/README.md`, an agent pointed at `"WH001": false` for that path and pushed back, and the reviewer recorded a *learning* never to flag WH001 there — the wrong invariant, learned off the wrong side of the contradiction ([#521](https://github.com/Wave-RF/WaveHouse/issues/521)). Both configs are deleted — each held nothing but the override, so the root `.markdownlint.json` governs again — and the 51 hard-wrapped paragraphs they were hiding are joined: 41 in `.github/workflows/README.md` and 10 in `.claude/skills/pm-triage/references/routine.md`, mechanical joins with no wording changed and every fenced block, table row, and heading byte-identical either side of the reflow. Deleted with them: the wizard-installed PostHog skill at `.claude/skills/integration-astro-view-transitions/` — 9 files, ~1,456 lines, including an 809-line `EXAMPLE.md` copied wholesale from `PostHog/context-mill`. Its integration job finished in [#277](https://github.com/Wave-RF/WaveHouse/pull/277), nothing in the repo calls it, and the docs-site setup it once described is documented where it belongs — in `docs/src/components/PostHog.astro` and this file. Keeping unowned third-party prose in the tree means content that drifts silently on every upstream bump and that nobody here reviews; it was also the single file that would have needed a special-case lint exclusion, so removing it is what lets WH001 apply with **no exception at all** rather than one documented one. Its two inventory rows in `claude-code.md` go with it, as does the now-dead `docs/posthog-setup-report.md` entry in the `scripts/docs-prose.sh` denylist (the wizard's other artifact, deleted back in [#502](https://github.com/Wave-RF/WaveHouse/pull/502)) and the copies of that denylist in `AGENTS.md` and `.github/prompts/docs-review.md`, which the script's header requires be kept in lockstep. Review of the change then turned up four more things the exclusion had been hiding, all fixed here: **WH001 has a blind spot** — `no-hard-wrapped-prose.mjs` classifies any line indented four or more spaces as an indented code block, so a *nested* list item is never joined, which left three hard-wrapped bullets in `.github/workflows/README.md` §"Adding a job" that the autofix could not see (unwrapped by hand; they were the last hard-wrapped prose paragraphs in the repo) and made `AGENTS.md`'s and `development.md`'s "a list item is joined as a unit" wrong for nested items (both now state the four-space caveat); the `scripts/docs-prose.sh` header told readers to keep its denylist in lockstep with **two** sibling copies when there are **three** — the missed one being `.claude/agents/docs-reviewer.md`, the gating subagent's own system prompt, which had in fact been silently out of sync for the whole life of the `posthog-setup-report.md` exclusion; the `.markdownlint-cli2.jsonc` header's "applies to every tracked Markdown file" was exact for WH001 but not WH002, which returns early on anything that isn't `.mdx`; and the job-graph diagram omitted `docs-deploy`'s `needs` edges from `unit`, `integration`, and `e2e`, contradicting invariant 2 three lines below it. The denylist also drops its `PERF-CLAIMS-REVIEW.md` entry — unlike the wizard artifact this one names a file that was **never tracked** at all, so it guarded a hypothetical; the list's other general cases are patterns (`*.draft.md`, `*.old.md`) that already cover a one-off review document, and a literal filename restated in four places is the outlier. `scripts/docs-prose.sh all` still resolves the same 27-file prose set. ### Removed -- **The Go-side validation, timestamp canonicalization, row-filter evaluation and compact encoding** (`internal/discovery/{validation,timestamp}.go`, `internal/policy/{rowfilter,numeric}.go`, `internal/ingest/compact.go`, `internal/api/ingest_seams.go`, `internal/stream/hub.go`): `discovery.Validate` and `CanonicalizeTimestamps`, `policy.RowVisible`, `ingest.EncodeCompactRow`, the `RecordValidator`/`InsertChecker` seams and `stream.NumericSpecOf` are gone. Their jobs — parsing a record, rendering a timestamp, deciding a row filter, writing the positional row — are ClickHouse's own now, through chtypes (see the type-layer entry in Changed). `typelayer.InsertSettings()` is the one static set of parsing settings the worker's `INSERT` and the ingest parse share. +- **The Go-side validation, timestamp canonicalization, row-filter evaluation and compact encoding** (`internal/discovery/{validation,timestamp}.go`, `internal/policy/{rowfilter,numeric}.go`, `internal/ingest/compact.go` and `internal/api/ingest_seams.go` (all deleted), `internal/stream/hub.go`): `discovery.Validate` and `CanonicalizeTimestamps`, `policy.RowVisible`, `ingest.EncodeCompactRow`, the `RecordValidator`/`InsertChecker` seams and `stream.NumericSpecOf` are gone. Their jobs — parsing a record, rendering a timestamp, deciding a row filter, writing the positional row — are ClickHouse's own now, through chtypes (see the type-layer entry in Changed). `typelayer.InsertSettings()` is the one static set of parsing settings the worker's `INSERT` and the ingest parse share. - **`policy.LiteralValue` and `policy.CanonicalNumericLiteral`** (`internal/policy/{canonical,policy}.go`): the marker type and the numeric re-reading of a policy-authored insert-check literal. Insert checks are chtypes filters now, so a literal binds as written and ClickHouse reads it under the column's type — there is no second, numeric reading at compare time. A `_eq: "1.0"` against a `UInt64` used to admit a stored `1`; the literal is no longer re-read numerically, so an insert check on it matches no row and the record is refused with `403` (the strict integer cast treats a non-canonical value as matching nothing), and the fix is to write a literal the column can read. `CanonicalScalar` stays: it is still the one rendering layer for a JWT claim. The released-version entry further down this file describing `LiteralValue` as shipped behaviour is left as history. @@ -116,7 +116,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Fixed -- **A timestamp filter on a column with a time zone matches the instant it names** (`internal/query/{builder,bind}.go` (`bind.go` new; + tests), `internal/api/{clickhouse_http,structured_query}.go` (+ tests, + integration-tagged `structured_query_integration_test.go`), `tests/e2e/sdk/query.test.ts`, `docs/src/content/docs/{api,architecture}.md`): `POST /v1/query` rewrote an RFC 3339 filter value, and formatted `time_range` bounds, as zone-less UTC text that ClickHouse then read in the **column's** zone — so on a `DateTime('Asia/Tokyo')` column `eq "2026-06-21T04:00:00Z"` matched nothing, on `DateTime64(3, 'America/New_York')` it matched the row four hours later, and on a zone-less column under a non-UTC server the row the server's offset away (measured on ClickHouse 26.8.15.10 and 24.8.14.39). WaveHouse no longer rewrites the value: on a `Date`/`DateTime`/`DateTime64` column it binds as written and ClickHouse parses it with `parseDateTime64BestEffort` in the column's declared zone, so an offset or `Z` is the exact instant, a zone-less value still means the column's local time, a fraction compares exactly against a whole-second column (ClickHouse refused it as a type mismatch before), and a value ClickHouse cannot parse is `400 clickhouse.rejected`. The primary key is still used. `time_range` bounds go as RFC 3339 in UTC through the same parse. **An `in` list is no longer capped**: it went as one query-string parameter, which ClickHouse caps at 128 KiB, and answered `400` past that; it now travels as an external table in a multipart body, so any list the 1 MiB request carries reaches ClickHouse, compared under the column's type and with the primary key in use. One difference from the old list: an element that is not a value of the column's type (`"1.5"` on an integer column) matches no row instead of failing the query. The limits left are ClickHouse's own, each a `400` that names it: one scalar value over 128 KiB once URL-encoded, all of them over the request line, and — for a query with an `in` list — SQL over the 128 KiB form field it travels in. +- **A timestamp filter on a column with a time zone matches the instant it names** (`internal/query/{builder,bind}.go` (`bind.go` new; + tests), `internal/api/{clickhouse_http,structured_query}.go` (+ tests, + integration-tagged `structured_query_integration_test.go`), `tests/e2e/sdk/query.test.ts`, `docs/src/content/docs/{api,architecture}.md`): `POST /v1/query` rewrote an RFC 3339 filter value, and formatted `time_range` bounds, as zone-less UTC text that ClickHouse then read in the **column's** zone — so on a `DateTime('Asia/Tokyo')` column `eq "2026-06-21T04:00:00Z"` matched nothing, on `DateTime64(3, 'America/New_York')` it matched the row four hours later, and on a zone-less column under a non-UTC server the row the server's offset away (measured on ClickHouse 26.8.15.10 and 24.8.14.39). WaveHouse no longer rewrites the value: on a `Date`/`DateTime`/`DateTime64` column it binds as written and ClickHouse parses it with `parseDateTime64BestEffort` in the column's declared zone, so an offset or `Z` is the exact instant, a zone-less value still means the column's local time, a fraction compares exactly against a whole-second column (before, 26.8 truncated it to the second and 24.8 refused it as a type mismatch; measured), and a value ClickHouse cannot parse is `400 clickhouse.rejected`. The primary key is still used. `time_range` bounds go as RFC 3339 in UTC through the same parse. **An `in` list is no longer bounded by the statement's size**: it was interpolated into the SQL text, which ClickHouse's `max_query_size` (256 KiB by default) bounds; it now travels as an external table in a multipart body, so any list the 1 MiB request carries reaches ClickHouse, compared under the column's type and with the primary key in use. One difference from the old list: an element that is not a value of the column's type (`"1.5"` on an integer column) matches no row instead of failing the query. The limits left are ClickHouse's own, each a `400` that names it: one scalar value over 128 KiB once URL-encoded, all of them over the request line, and — for a query with an `in` list — SQL over the 128 KiB form field it travels in. - **A schema refresh that started first can no longer publish last** (`internal/discovery/discovery.go` (+ tests), `tests/integration/setup_test.go`, `AGENTS.md`, `docs/src/content/docs/architecture.md`): overlapping refreshes of one tenant, such as `POST /v1/ops/schema/refresh` during an auto-refresh tick, published in the order they finished, so a refresh that read `system.columns` before a table was created could land after one that saw it and drop the table from the registry and the type layer until the next refresh. Each refresh now takes a generation before its first read and publishes, and runs its hooks, only if it started after the refresh whose snapshot is published; one that lost the race returns success without publishing. - **A tenant moved to another ClickHouse address or database discovers its new schema with the reload** (`internal/app/{wire,discoveries}.go` (+ tests), `internal/discovery/discovery.go` (comment), `internal/testutil/testutil.go`, `docs/src/content/docs/architecture.md`, `settings-directory.mdx`, `AGENTS.md`): closes [#638](https://github.com/Wave-RF/WaveHouse/issues/638). A reload that changed a tenant's `clickhouse.addr` or `clickhouse.database` orphaned the tenant's cache and left its schema registry as it was, so until the tenant's loop fired at `schema.refresh_interval` its queries and inserts were validated against the previous database's schema and run against the new one. The tenants the pools reconcile reports stale now have their registry dropped in the same hook, and the discovery reconcile that follows builds each a fresh one over the pool it is on, as it does for a tenant back after a rejection or removal: the first discovery runs at once in the tenant's own loop, so the reload never waits on ClickHouse, and until it succeeds the tenant's table lookups answer `503` with `Retry-After: 5`, as before any first discovery. A discovery that fails is logged with its tenant (`schema discovery retry failed`), counted in `wavehouse_schema_refresh_failures_total`, and retried with backoff from two seconds to sixty. A tenant whose address and database did not change keeps its registry and its loop, a flat directory's tenant `0` moves the same way, and a process without the api role, which discovers no schema, only repoints its pool. @@ -126,9 +126,9 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Dedupe claims an id, publishes, then commits it — and keys it by tenant, table and id** (`internal/dedupe/{dedupe,key,embedded,managed}.go` (+ tests), `internal/dedupe/dedupetest/` (new), `internal/api/ingest.go` (+ tests), `internal/settings/validate_test.go`, `internal/keyenc/keyenc.go`, `internal/testutil/mocks.go`, `internal/app/app_test.go`, `AGENTS.md`, `docs/src/content/docs/{architecture,api,deployment,development}.md`, `settings-directory.mdx`, `sdk/reference.md`): `CheckAndMark` is replaced by a two-phase `Reserve` → `Commit` / `Release` contract with a lease on the pending claim, and every backend now runs one conformance suite. Four bugs go with it. Two concurrent requests carrying one id no longer both publish it: Pebble's check and claim happen under one lock, and the loser answers `503` with `Retry-After` while the winner is still publishing ([#390](https://github.com/Wave-RF/WaveHouse/issues/390)). A publish that fails gives its id back, so the retry a `503` asks for is published instead of skipped as a duplicate of a record that never reached the queue ([#384](https://github.com/Wave-RF/WaveHouse/issues/384)) — the residual case is a publish that fails after it already reached the broker (a timeout, a disconnect), where the released id lets the retry through but that retry publishes a genuine second copy; [#629](https://github.com/Wave-RF/WaveHouse/pull/629) closes that with an idempotency key. The same id in two tables is two ids ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s keyspace half). An explicit `null` id is a missing id — rejected under `require_id`, published un-deduped otherwise — instead of the one id `""` that made every null record after the first a duplicate ([#370](https://github.com/Wave-RF/WaveHouse/issues/370)). **Upgrade:** the key layout changes, so an id seen before the upgrade is accepted once more after it; nothing is migrated, and the old keys are left in `/pebble`, unread, deleted by the retention sweep (see Added) ([#220](https://github.com/Wave-RF/WaveHouse/issues/220)) ([Deployment → Upgrading across the dedupe key change](https://github.com/Wave-RF/WaveHouse/blob/main/docs/src/content/docs/deployment.md#upgrading-across-the-dedupe-key-change)). The key is readable text, `/
/` (for example `acme/clicks/evt-123`), with the table and id escaped and joined by `internal/keyenc`, the escaping NATS subject tokens already use, so any table name gets a keyspace of its own, including one holding a NUL byte or a `/`. New metrics: `wavehouse_ingest_dedupe_commit_failed_total` (a published record whose id failed to commit; the claim lapses with its lease) and `wavehouse_dedupe_hashed_id_total` (an id over 1,024 bytes once escaped, stored as its SHA-256). - **Ingest runs in windows of 256 records over the dedupe contract, and a dedupe store that cannot answer is a `503`** (`internal/api/ingest.go` (+ tests), `internal/mq/{mq,embedded}.go` (+ tests), `internal/dedupe/key.go` (+ tests), `internal/testutil/mocks.go`, `AGENTS.md`, `docs/src/content/docs/{api,architecture,durability}.md`, `settings-directory.mdx`, `sdk/reference.md`): each window of a request is prepared, then reserved in one dedupe call, published in order, and committed in one call, so a batch costs one dedupe round trip per phase per window rather than per record — on Pebble, one commit `fsync` per window (a 1,000-record batch: four syncs instead of a thousand, 24 ms against 5.7 s of dedupe time measured with the queue stubbed). Every deduped record is published under an idempotency key (`mq.WithIdempotencyKey`, JetStream's message id, derived by `dedupe.IdempotencyKey`), and each tenant's ingest stream now keeps an explicit two-minute duplicate window, sized to `2 × the 30-second lease + 1s`: an uncertain publish's `503` sends the *full* lease as `Retry-After`, so an obedient client's retry can land up to ~2×lease after the original request, and the `+1s` covers a backend whose claim expiry itself rounds up by that much. That closes the last path of [#384](https://github.com/Wave-RF/WaveHouse/issues/384): a publish that fails with an unknown outcome (anything but a full queue) keeps its record's claim until the lease lapses instead of releasing it, and a retry after the lease but within two minutes of the first publish is dropped by the queue if the first copy was stored (a later one is stored again). A dedupe store that is not open or that reports `dedupe.ErrUnavailable` now answers `503 {"error":"dedupe store unavailable"}` with `Retry-After: 5`, which the SDK retries, rather than `500 dedupe failed`. A mid-body read error or dedupe failure now drops the open window unpublished, where records before it used to be published; `wavehouse_ingest_dedupe_commit_failed_total` and `wavehouse_ingest_dedupe_disabled_total` count records, as before, now added a window at a time. - **Tests that start the embedded broker no longer fail removing its store after passing** (`internal/testutil/storedir` (new, + tests), `internal/testutil/testutil.go`, `internal/mq/embedded.go` (comment), `internal/mq/{embedded_test,mqtest/embedded_test}.go`, `internal/ingest/worker_test.go`, `internal/app/{app,roles}_test.go`, `cmd/wavehouse/main_test.go`, `tests/integration/{ingest_outage,query_errors,tenants}_test.go`, `AGENTS.md`, `docs/src/content/docs/development.md`): [#442](https://github.com/Wave-RF/WaveHouse/issues/442). The NATS server writes each durable consumer's state (`obs//o.dat`, through a temporary file renamed into place) from a goroutine that neither `Shutdown` nor `WaitForShutdown` joins, and its consumer store waits for that goroutine at close only when state is still unwritten, for at most 100ms — so a write already under way lands after `EmbeddedNATS.Close` returns, and `t.TempDir`'s one-shot `RemoveAll` met the late entry as `directory not empty`. Under parallel test processes it failed about 4% of the ingest worker tests (78 of 1,800 runs). Every store a test puts on disk now comes from `storedir.New(t)`, whose cleanup — after the broker's `Close` — removes it again whenever a directory was refilled between being read and being removed: each late write adds at most two entries and none once its directory is gone, so the removal ends without a timer (0 of 1,800 under the same load). It replaces two sleep-and-retry copies in the `internal/mq` tests. `TestStartIngestWorker_StopFunc_RespectsShutdownDeadline` also joins the worker its deadline abandons before the broker closes, rather than leaving it to ack on a closed connection. -- **A pipe that writes runs on every call instead of being answered from the cache** (`internal/api/{pipes,ch_errors}.go` (+ tests), `docs/src/content/docs/{pipes.mdx,api.md,architecture.md,configuration.mdx,settings-directory.mdx,ingest-pipeline.md,sdk/pipes.md,sdk/reference.md}`, `clients/ts/src/pipes.ts` (doc comment), `internal/{settings/settings,app/wire}.go` (comments), `AGENTS.md`): fixes [#386](https://github.com/Wave-RF/WaveHouse/issues/386). `/v1/pipes/{name}` sent a write's SQL to ClickHouse through `Exec`, but still cached the `[]` it returned and coalesced identical calls in flight, so a repeat within the TTL answered `200` without executing and concurrent identical calls became one write — silently dropped writes, and with a shared cache ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)) on every instance. A pipe whose bound SQL `IsMutation` classifies as a write — the same classifier that picks `Exec` — now skips the cache lookup, the fill and singleflight, and answers `X-Cache: BYPASS` with `Cache-Control: no-store`, so an HTTP cache in front of a `GET` cannot drop the write either. Classification stays automatic rather than a declared pipe property, so an operator cannot forget to mark one, and costs no ClickHouse round trip. A failed write answers with the status and `code` a failed read gets (see the ClickHouse-errors entry below), but always `retryable: false` and with no `Retry-After`, `503 clickhouse.unavailable` included: the statement may have run, so the SDK does not retry it. A write refused before it is sent, the tenant on no pool, keeps its `503` with `Retry-After: 30`. Read pipes are unchanged. Not in this fix: a write pipe still does not invalidate cached reads of the table it writes ([#394](https://github.com/Wave-RF/WaveHouse/issues/394)). +- **A pipe that writes runs on every call instead of being answered from the cache** (`internal/api/{pipes,ch_errors}.go` (+ tests), `docs/src/content/docs/{pipes.mdx,api.md,architecture.md,configuration.mdx,settings-directory.mdx,ingest-pipeline.md,sdk/pipes.md,sdk/reference.md}`, `clients/ts/src/pipes.ts` (doc comment), `internal/{settings/settings,app/wire}.go` (comments), `AGENTS.md`): fixes [#386](https://github.com/Wave-RF/WaveHouse/issues/386). `/v1/pipes/{name}` sent a write's SQL to ClickHouse through `Exec`, but still cached the `[]` it returned and coalesced identical calls in flight, so a repeat within the TTL answered `200` without executing and concurrent identical calls became one write — silently dropped writes, and with a shared cache ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)) on every instance. A pipe whose bound SQL `IsMutation` classifies as a write — the same classifier that decides how the statement is sent — now skips the cache lookup, the fill and singleflight, and answers `X-Cache: BYPASS` with `Cache-Control: no-store`, so an HTTP cache in front of a `GET` cannot drop the write either. Classification stays automatic rather than a declared pipe property, so an operator cannot forget to mark one, and costs no ClickHouse round trip. A failed write answers with the status and `code` a failed read gets (see the ClickHouse-errors entry below), but always `retryable: false` and with no `Retry-After`, `503 clickhouse.unavailable` included: the statement may have run, so the SDK does not retry it. A write refused before it is sent, the tenant on no pool, keeps its `503` with `Retry-After: 30`. Read pipes are unchanged. Not in this fix: a write pipe still does not invalidate cached reads of the table it writes ([#394](https://github.com/Wave-RF/WaveHouse/issues/394)). - **The write classifier skips whitespace, comments and quoted text the way ClickHouse's lexer does, classifies a `WITH`-led statement by `INSERT INTO` alone, and looks through `EXECUTE AS`** (`internal/api/clickhouse_exec.go` (+ tests), `internal/testutil/mutationtest` (new), `tests/integration/ismutation_test.go` (new), `docs/src/content/docs/pipes.mdx`, `AGENTS.md`): `IsMutation` picks `Exec` for a write, and since [#386](https://github.com/Wave-RF/WaveHouse/issues/386) keeps a write pipe out of the cache. It missed a write behind a backslash-escaped quote (`'it\'s'`, and the same inside `"…"` and `` `…` ``), a heredoc (`$$ ( $$`, `$tag$ … $tag$`), a curly-quoted literal or identifier (`‘(’`, `“c(d”`), a `//` line comment, a nested block comment (`/* a /* b */ SELECT */ INSERT …`), a number led by `.` with the verb glued to it (`WITH 1 AS a, .5INSERT INTO t …`, which ClickHouse reads as `.5` then `INSERT`), an `EXECUTE AS ` prefix (`EXECUTE AS u INSERT …`), or leading whitespace other than space, tab, CR and LF: `\v`, `\f`, a no-break space, a byte-order mark, and the other Unicode spaces ClickHouse skips. A missed write went through `Query`, which ran it and then failed the call with a `5xx` the TypeScript SDK retries, so one call could write three times. The same gaps, and a word led by `_` (`_delete`) whose tail was read as a verb, could make a read look like a write, which runs through `Exec` and answers `[]`. After a `WITH` list, which ClickHouse follows only with `SELECT`, a FROM-first `SELECT` or `INSERT INTO`, a name spelled like a keyword was taken for the statement: `WITH 'd' AS desc INSERT …` and `WITH 1 AS select INSERT …` ran as reads, and `WITH 1 AS set SELECT set` and `WITH 1 AS x FROM system.one SELECT x` as writes. A `WITH`-led statement is now a write exactly when it holds `INSERT INTO` outside parentheses. The classifier, exported as `IsMutation` for it, is now checked against the pinned ClickHouse's own parser (`EXPLAIN AST`) in the integration suite: every test case, and every ClickHouse keyword as a `WITH` list's name ahead of each statement a `WITH` list can lead. -- **A failed ClickHouse query answers by what went wrong, not a flat `500`/`502`** (`internal/api/ch_errors.go` (new, + tests), `internal/api/{errors,query,structured_query,pipes,schema,ch_settings}.go`, `internal/chconn/errclass.go` (`HTTPStatus` exported), `clients/ts/src/errors.ts` (+ tests), `tests/integration/query_errors_test.go` (new), `tests/integration/query_limits_test.go`, `internal/app/app_test.go`, `tests/e2e/sdk/{admin,query}.test.ts`, `AGENTS.md`, `docs/src/content/docs/{api,architecture}.md`, `docs/src/content/docs/{access-control,configuration}.mdx`, `docs/src/content/docs/sdk/{reference.md,index.mdx}`): fixes [#403](https://github.com/Wave-RF/WaveHouse/issues/403) and [#271](https://github.com/Wave-RF/WaveHouse/issues/271), part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). ClickHouse answers a syntax error, a missing grant and an overloaded server alike with HTTP `500`, so `/v1/ops/query` turned a bad statement into a `502` and `/v1/query` and pipes into a `500` the SDK retried. All three now class the failure with `chconn.Classify` through one helper, `writeCHError`: a statement ClickHouse refused is `400 clickhouse.rejected`; a query over a rows/bytes limit, the role's own memory cap, or its time cap where that is no longer than `query_timeout` is `400 clickhouse.limit_exceeded`; `ACCESS_DENIED` is `403 clickhouse.access_denied`; credentials, user or database refused, or a redirect or `4xx` with no exception code from whatever fronts ClickHouse, is `502 clickhouse.misconfigured`; ClickHouse down, unreachable or overloaded is `503 clickhouse.unavailable` with `Retry-After: 5`; a failure with no verdict stays `500` (`502` on the proxy) as `clickhouse.unknown`. The error envelope gains `code` and `retryable` next to `error` on these responses — additive. A role with `max_execution_time` now queries with no context deadline and a cancel two seconds past the cap instead: clickhouse-go overwrote the cap's `max_execution_time` with deadline+5s for any deadline over 1s, so an overrun came back as a bare deadline, indistinguishable from waiting for a pooled connection; ClickHouse now enforces the cap itself and reports `TIMEOUT_EXCEEDED`. `POST /v1/ops/schema/refresh` against an unreachable ClickHouse is a `503` with `Retry-After` instead of a `500`. **SDK:** `WaveHouseError.code` and `retryable` now take the server's `code`/`retryable` when the body has them (`HTTP_` and "5xx retries" otherwise), so a rejected query is `clickhouse.rejected` rather than `HTTP_500`, and is not retried. +- **A failed ClickHouse query answers by what went wrong, not a flat `500`/`502`** (`internal/api/ch_errors.go` (new, + tests), `internal/api/{errors,query,structured_query,pipes,schema,ch_settings}.go`, `internal/chconn/errclass.go` (`HTTPStatus` exported), `clients/ts/src/errors.ts` (+ tests), `tests/integration/query_errors_test.go` (new), `tests/integration/query_limits_test.go`, `internal/app/app_test.go`, `tests/e2e/sdk/{admin,query}.test.ts`, `AGENTS.md`, `docs/src/content/docs/{api,architecture}.md`, `docs/src/content/docs/{access-control,configuration}.mdx`, `docs/src/content/docs/sdk/{reference.md,index.mdx}`): fixes [#403](https://github.com/Wave-RF/WaveHouse/issues/403) and [#271](https://github.com/Wave-RF/WaveHouse/issues/271), part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). ClickHouse answers a syntax error, a missing grant and an overloaded server alike with HTTP `500`, so `/v1/ops/query` turned a bad statement into a `502` and `/v1/query` and pipes into a `500` the SDK retried. All three now class the failure with `chconn.Classify` through one helper, `writeCHError`: a statement ClickHouse refused is `400 clickhouse.rejected`; a query over a rows/bytes limit, the role's own memory cap, or its time cap where that is no longer than `query_timeout` is `400 clickhouse.limit_exceeded`; `ACCESS_DENIED` is `403 clickhouse.access_denied`; credentials, user or database refused, or a redirect or `4xx` with no exception code from whatever fronts ClickHouse, is `502 clickhouse.misconfigured`; ClickHouse down, unreachable or overloaded is `503 clickhouse.unavailable` with `Retry-After: 5`; a failure with no verdict stays `500` (`502` on the proxy) as `clickhouse.unknown`. The error envelope gains `code` and `retryable` next to `error` on these responses — additive. A role with `max_execution_time` now queries with no context deadline and a cancel two seconds past the cap instead: clickhouse-go overwrote the cap's `max_execution_time` with deadline+5s for any deadline over 1s, so an overrun came back as a bare deadline, indistinguishable from waiting for a pooled connection; ClickHouse now enforces the cap itself and reports `TIMEOUT_EXCEEDED` (superseded in part: `/v1/query` and pipes now run over HTTP — see the structured-query entry — and still send the cap as `max_execution_time`, giving up two seconds past it; clickhouse-go is no longer on those paths). `POST /v1/ops/schema/refresh` against an unreachable ClickHouse is a `503` with `Retry-After` instead of a `500`. **SDK:** `WaveHouseError.code` and `retryable` now take the server's `code`/`retryable` when the body has them (`HTTP_` and "5xx retries" otherwise), so a rejected query is `clickhouse.rejected` rather than `HTTP_500`, and is not retried. - **An unavailable ClickHouse is retried with backoff instead of dead-lettering every row** (`internal/chconn/errclass.go` (new, + tests), `internal/ingest/{worker,backoff}.go` (`backoff.go` new, + tests), `internal/mq/{mq,embedded}.go`, `internal/testutil/mocks.go`, `tests/integration/ingest_outage_test.go` (new), `AGENTS.md`, `README.md`, `docs/src/content/docs/{ingest-pipeline,architecture,api,deployment,why-wavehouse}.md`, `docs/src/content/docs/{settings-directory,index,access-control}.mdx`): workstream A of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). A failed batch insert used to go through row-by-row isolation whatever the failure, so a ClickHouse that was down, overloaded or read-only failed every row twice and parked the whole batch on the DLQ. `chconn.Classify` now classes the failure first — `Rejected` (any ClickHouse exception code outside the availability and credential lists: the server read the row and refused it), `Unavailable` (connection refused/reset, timeouts, `TOO_MANY_SIMULTANEOUS_QUERIES`, `SERVER_OVERLOADED`, `MEMORY_LIMIT_EXCEEDED`, `TOO_MANY_PARTS`, `READONLY`, `TABLE_IS_READ_ONLY`, `KEEPER_EXCEPTION`, …), `Denied` (`AUTHENTICATION_FAILED`, `ACCESS_DENIED`, …) or `Unknown` (no code, no recognizable transport failure). Only `Rejected` is isolated and dead-lettered as before, and a multi-row batch refused with `TOO_MANY_PARTS` or `MEMORY_LIMIT_EXCEEDED` is split row by row first (`chconn.Splittable`), because a batch spanning too many partitions or too much memory can fail where each of its rows inserts; every other class hands the batch back to the queue with a delayed nak (`mq.Message.NakWithDelay`, new) under a jittered 1 s → 30 s backoff shared by every table on the same ClickHouse pool (a failure of one table — read-only, too many parts or mutations, a grant missing on it, `chconn.TableScoped` — backs off that table alone), which turns rows away without a request while it runs and probes once per window, and ClickHouse going away mid-isolation stops isolation and retries the rows it had not settled. Counted by the new `wavehouse_ingest_retries_total{table, reason}`; logged at `WARN` when an outage starts and at most every 30 s during it. A long outage now shows as a growing ingest stream and, at `mq.max_bytes_gb`, ingest `503`s — not as a full DLQ; a lasting failure of one table holds back its tenant's other tables once its waiting rows reach `maxAckPending`. Retried rows come back out of arrival order, which matters only to a `ReplacingMergeTree` without a version column or a `CollapsingMergeTree`. - **Schema discovery's retry loop jitters its backoff** (`internal/discovery/discovery.go` (+ tests), `internal/app/wire.go`, `internal/api/errors.go`, `AGENTS.md`, `docs/src/content/docs/{architecture,api,deployment}.md`): `RetryRefresh` slept exactly `2s * 2^n` capped at 60s, so instances retrying against one recovering ClickHouse fired in lockstep, every 60s on the same second. Each sleep is now drawn uniformly from below the backoff (full jitter), spreading the retries over the whole window and halving the mean wait — so a failing tenant's retries, their log lines and `wavehouse_schema_refresh_failures_total` come about twice as often ([#141](https://github.com/Wave-RF/WaveHouse/issues/141)). - **A write that lands while a cached read is running no longer re-homes the pre-write rows under the post-write key** (`internal/cache/{cache,local,version_manager}.go` (+ tests), `internal/testutil/cachetest` (new), `internal/api/{structured_query,pipes}.go` (+ tests), `internal/ingest/worker.go` (+ tests), `internal/query/ident.go` (removed, + tests), `internal/app/wire.go`, `docs/src/content/docs/{api,architecture,deployment}.md`, `AGENTS.md`): fixes [#382](https://github.com/Wave-RF/WaveHouse/issues/382), part of [#613](https://github.com/Wave-RF/WaveHouse/issues/613). `POST /v1/query` rebuilt the version-folded cache key after the query ran, so an insert invalidating the table mid-query filed the rows read before it under the new versions, and they were served as fresh until their TTL (a pipe result's key folded no version, so pipes were unaffected; with the tenant's version in every key they now take the same snapshot). The `Cache` interface now snapshots at lookup: `Lookup(ctx, tenant, sha, deps)` returns the `Entry` and a `Snapshot` of the versions it read, and `Set(ctx, snapshot, value, ttl)` stores under that snapshot, so such a fill is orphaned and the next request reads the post-write rows. The singleflight leader's snapshot is the one used; coalescing is unchanged. The snapshot is taken before any input a bump invalidates is chosen, the tenant's ClickHouse connection included: both handlers now look up before they resolve the tenant's pool, so a reload that moves the tenant to another address or database after a request took the old pool orphans that request's fill instead of filing the old database's rows as fresh under the new tenant version. A tenant on no pool is still a `503` before a cached result is served or a query runs. A `Lookup` whose dependencies name another tenant is refused (`ErrForeignDependency`). `Set` now errors only when the backend failed: a value the cache declines (larger than the cache holds (`cache.l1_max_cost`), a non-positive TTL) is not an error. One behavior change: the tenant's version is folded into every key, a pipe result's included, so `InvalidateTenant` (a tenant back on a pool after an absence, or moved to another ClickHouse address or database) now drops that tenant's cached pipe results as well as its query results; before, a pipe result stayed until its TTL. Inserts still do not invalidate pipe results ([#343](https://github.com/Wave-RF/WaveHouse/pull/343)). A backend-agnostic conformance suite, `cachetest.Run`, pins what a hit, a miss and each kind of bump mean, and `LocalCache` runs it; the Redis-compatible backend will run the same suite. The cache now escapes table and scope names itself: a `Namespace` carries them raw and the cache renders each result's key with `internal/keyenc`, so neither the structured-query read nor the ingest worker's invalidation escapes them (`query.SafeEncodeToken` is gone) and no name reaches a key unescaped. The suite pins that a name holding a dot, a space or a `%` is read and bumped under one key, and that names which would run together unescaped (`a.0.b` against `a` with scope `b.0.`) stay two entries. The version index's own entries are unaffected by the escaping; only the rendered key changes: it now carries the tenant's version and escapes the caller's query key whole. All of them live in the process, so nothing stored is orphaned. @@ -139,7 +139,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The `?token=` strip no longer repairs a query string that does not parse** (`internal/auth/auth.go`): `bearerToken` removed a query-string token by parsing the query, deleting `token`, and re-encoding what was left — and `url.ParseQuery` skips a pair it cannot read, so the re-encoding erased that pair. A handler that parses the query strictly in order to refuse a malformed one would then see a clean query: `GET /v1/ops/pipes?tenant=acme;x=1&token=…` would have answered `200` with the default tenant's pipes. The token is read exactly as before and a query that parses is rewritten exactly as before; a query that does not parse now loses its token pairs and nothing else, byte for byte. Pinned through `api.NewRouter` with the real authenticator, since a handler-level test never runs the middleware that rewrote the URL. - **SSE gap-fill re-reads the policy per replayed row** (`internal/stream/hub.go`): `ReplayProjector` captured the policy once when the replay began, so a policy adopted mid-fill — a revoked grant, say — applied only after the fill ended. It now reads it per event, as `Broadcast` does on the live path. -- **A row-filter claim containing a backslash, tab or newline no longer withholds rows the query path returns** (`internal/chsql/chsql.go`, `internal/chsql/chsql_test.go`, `internal/query/builder.go`, `internal/typelayer/filter.go`, `internal/typelayer/filter_test.go`, `tests/integration/rowfilter_stream_test.go`): ClickHouse reads a scalar `{p:String}` parameter with its escaped-text reader, so an unescaped `a\b` arrived holding a backspace and a raw tab or newline was a hard parse error. The stream compared such a claim false and the query path could misread it. Both surfaces now encode `\`, tab, newline and carriage return through one `chsql.EscapeStringParam`, measured byte-for-byte against ClickHouse 26.6.3.62 and against the chtypes artifact. The old behaviour was fail-closed — rows withheld, never leaked — so this is availability, not confidentiality. +- **A row-filter claim containing a backslash, tab or newline no longer withholds rows the query path returns** (`internal/chsql/chsql.go`, `internal/chsql/chsql_test.go`, `internal/query/{builder,bind}.go`, `internal/typelayer/filter.go`, `internal/typelayer/filter_test.go`, `tests/integration/rowfilter_stream_test.go`): fixes the type-layer and structured-query entries above before they ship. ClickHouse reads a scalar `{p:String}` parameter with its escaped-text reader, so an unescaped `a\b` arrived holding a backspace and a raw tab or newline was a hard parse error. The stream compared such a claim false and the query path could misread it. Both surfaces now encode `\`, tab, newline and carriage return through one `chsql.EscapeStringParam`, measured byte-for-byte against ClickHouse 26.6.3.62 and against the chtypes artifact. The unfixed behaviour (never released) was fail-closed — rows withheld, never leaked — so this is availability, not confidentiality. - **`classify-paths.sh` no longer reads a `grep` failure as "no match"** (`scripts/classify-paths.sh`, `scripts/classify-paths.test.sh`): both decisions were `if printf … | grep -qE …; then A; else B; fi`. `grep` exits `0` on match, `1` on no match and **`2` on error** (can't fork/exec, read error, bad pattern), and the `else` branch collapsed `1` and `2` into the same answer — `set -euo pipefail` does not help, since `set -e` is suppressed for a command used as an `if` condition. Observed twice in local `make ci` runs whose static checks run at `-j 14`: a different single case failed each time (`mixed-docs-go` answering `docs=false`, then `dep-bump-go` answering `code=false`) while every other case passed, which is the signature of a transient `grep` failure rather than a pattern bug. The test caught it only because it asserts expected values; **the production path has no such check** — CI's `changes` job gates the docs pipeline on this answer, so a `docs=false` produced by an errored `grep` silently skips the docs build and still reports success. The two greps now go through a `matches` helper that aborts with a diagnostic on any exit above 1, and the test suite stubs `grep` onto `PATH` to prove the abort fires (that case fails against the previous script). A second instance of the same class, found reviewing the first fix: the helper piped its input into `grep -q`, which exits at the first match — so once the file list outgrew the pipe buffer (a few thousand paths) the upstream `printf` died of SIGPIPE, `pipefail` reported 141, and the new error arm aborted on an ordinary large change set. Reproduced at 5,000 paths. It now reads from a here-string instead, and the suite pins that case. `scripts/ci/classify-changes.sh` also stopped reading the classifier through process substitution, which discarded its exit status: a classifier that aborted left `code`/`docs` empty, every `needs.changes.outputs.code == 'true'` job skipped, and the `CI` aggregator reported green having run nothing. It now captures the status, and fails closed — running everything — on a failed *or* partial classification, matching the rule already used for an empty file list. Also here, unrelated and one line: `biome.json` declared `$schema` 2.4.15 while the lockfile pins the 2.5.8 CLI, so `biome check --error-on-warnings` failed on the config itself for any change touching TypeScript. Bumped to match; it changes no lint rule. @@ -154,7 +154,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The pipes page no longer says a parameter can never break out of its literal** (`docs/src/content/docs/pipes.mdx`, `internal/pipes/pipes.go`): that holds only for a placeholder written bare. A string value brings its own quotes, so inside a quoted placeholder they close the template's: the body `{"id": " OR 1=1 OR id = "}` turns `WHERE id = '{{id}}'` into `WHERE id = '' OR 1=1 OR id = ''`, which matches every row. The page now says to write each placeholder bare, never inside quotes. Check existing `pipes.json` templates for quoted placeholders (`'{{x}}'`) and write them bare ([#662](https://github.com/Wave-RF/WaveHouse/issues/662)). - **An empty HMAC secret no longer verifies tokens signed with an empty key** (`internal/auth/auth.go` (+ tests), `SECURITY.md`): with `auth.jwt_secret` unset and no `auth.jwks_url` — the documented public-access posture, "no token can validate" — the key function handed `golang-jwt` an empty HMAC key, and the library verifies a token signed with one, so anyone could mint `{"role": "admin"}` and reach the whole data plane and `/v1/ops/*`. The verifier now refuses every token when it has neither a secret nor a JWKS URL, pinned by a test that signs with the empty key. Found by review on [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9 ([#604](https://github.com/Wave-RF/WaveHouse/pull/604)), which carries the same fix. -- **A policy claim compared against an integer column goes through a strict cast, so a claim that does not fit the column matches nothing instead of wrapping** (`internal/chsql/chsql.go`, `internal/policy/policy.go`, `internal/query/builder.go`, `internal/typelayer/{filter,typelayer}.go`, `docs/src/content/docs/{access-control.mdx,api.md,architecture.md}`, + tests): a claim bound as a plain `{p:String}` wrapped modulo 2^64 on every integer column (and at their own width on `[U]Int128`/`[U]Int256`), identically on `/v1/query`, the stream and the insert check, so a claim of `18446744073709551621` read and wrote tenant `5`. Row filters and insert checks now compare an integer column (`Nullable`/`LowCardinality` included) against `if(toString(accurateCastOrNull({p:String}, 'T')) = {p:String}, accurateCastOrNull({p:String}, 'T'), NULL)`: an in-range canonical claim answers exactly as before and keeps the primary key in use, while an out-of-range or non-canonical one (`007`, `+5`, `1.0`) matches no row on any operator and an insert check refuses it with `403` (a non-canonical claim used to be a per-row code 53 — no rows on a read, `422` on an insert). Other column types and a caller's own filters keep the plain form. +- **A policy claim compared against an integer column goes through a strict cast, so a claim that does not fit the column matches nothing instead of wrapping** (`internal/chsql/chsql.go`, `internal/policy/policy.go`, `internal/query/builder.go`, `internal/typelayer/{filter,typelayer}.go`, `docs/src/content/docs/{access-control.mdx,api.md,architecture.md}`, + tests): a claim bound as a plain `{p:String}` wrapped modulo 2^64 on every integer column (and at their own width on `[U]Int128`/`[U]Int256`), identically on `/v1/query`, the stream and the insert check, so a claim of `18446744073709551621` read and wrote tenant `5`. Row filters and insert checks now compare an integer column (`Nullable`/`LowCardinality` included) against `if(toString(accurateCastOrNull({p:String}, 'T')) = {p:String}, accurateCastOrNull({p:String}, 'T'), NULL)`: an in-range canonical claim answers exactly as before and keeps the primary key in use, while an out-of-range or non-canonical one (`007`, `+5`, `1.0`) matches no row on any operator and an insert check refuses it with `403` (a non-canonical claim used to fail: the whole `/v1/query` read with code 53 — `400 clickhouse.rejected` — a withheld stream row (reason `error`), and a `422` on an insert). Other column types and a caller's own filters keep the plain form. - **Policy validation now rejects the fail-open rule shapes strict decoding can't see** (`internal/policy/policy.go`, `docs/src/content/docs/access-control.mdx`; closes [#460](https://github.com/Wave-RF/WaveHouse/issues/460)): four new `validateRolePerms` rejections close the fail-open shapes strict decoding can't see because the document is syntactically innocent. A `filter` entry with no operator (`"tenant_id": {}`) resolved to zero predicates — no `WHERE` clause, row security silently off, the same shape a misspelled `"eq"` for `"_eq"` used to decode to before strict decoding closed that route; it is now rejected, as is its check-path twin (an operator-less `check` entry, skipped by `Evaluate`'s resolve switch — accepted but constraining nothing) and `filter:` under an `insert:` grant (resolved and then ignored by the ingest path — the same accept-but-ignore family as [#224](https://github.com/Wave-RF/WaveHouse/issues/224), and the pointed asymmetry #460 called out against the loud `check` `_neq`/`_gt`/`_lt` rejection) along with its mirror, `check:` under a `select:` grant — the likelier authoring slip and the fail-open direction: the author believes reads are row-scoped while `Evaluate` resolves the entry and nothing on the select or stream paths reads it. Because [#508](https://github.com/Wave-RF/WaveHouse/pull/508) funneled every adoption through the one `policy.Validate` path, the four checks land on boot, the directory watch, `SIGHUP`, `POST /v1/ops/settings/reload`, and `wavehouse validate` at once. #460's migration caveat (a stored policy hard-failing at boot) has evaporated with the settings directory being new and unreleased; no shipped seed, compose, or fixture policy carries any of the rejected shapes. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 8ca9c126..7ba387cd 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -34,12 +34,12 @@ Open a [feature request issue](https://github.com/Wave-RF/WaveHouse/issues/new?t 1. Ensure your changes pass all checks: ```bash - make ci # full local pipeline: verify + builds + all test suites (Docker required) + make ci # full local pipeline: verify + builds + all test suites (Docker and the chtypes artifact required) ``` - The pre-push hook (installed by `make tools`) blocks a push until the tree has been validated locally: a code change needs `make ci`, a docs/prose-only change needs only `make verify` (the same split CI makes). `make lint` / `make test` / `make build` are fast inner-loop subsets. + `make ci` needs the chtypes artifact: run `scripts/fetch-chtypes.sh` once per machine (see the [Development Guide](https://wavehouse.dev/development#the-chtypes-artifact--fetch-it-once-per-machine)). The pre-push hook (installed by `make tools`) blocks a push until the tree has been validated locally: a code change needs `make ci`, a docs/prose-only change needs only `make verify` (the same split CI makes). `make lint` / `make test` / `make build` are fast inner-loop subsets. -2. Write tests for new functionality. Unit tests go alongside the code in `internal/`. Integration tests carry the `//go:build integration` tag and go in `tests/integration/`, or beside the package when they test one package against its own external server (e.g. `internal/cache/redis_integration_test.go`), with the package added to the `test-integration` target. A test that must import NATS, which only `internal/mq` may do, goes in `internal/mq/natsspike`, or in `internal/mq` itself with the `integration` tag when it needs the package's internals (the external NATS broker's tests, which `make test-integration` selects by name). +2. Write tests for new functionality. Unit tests go alongside the code in `internal/`. Integration tests carry the `//go:build integration` tag and go in `tests/integration/`, or beside the package when they test one package against its own external server (e.g. `internal/cache/redis_integration_test.go`), with the package added to the `test-integration` target. A test that must import NATS, which only `internal/mq` may do, goes in `internal/mq/natsspike`, or in `internal/mq` itself with the `integration` tag when it needs the package's internals (the external NATS broker's tests); a test elsewhere that only needs a running NATS server uses `internal/mq/natstest`. In `internal/mq` and `internal/api` the `test-integration` target selects tagged tests by name, so match its `-run` pattern (`TestIntegration_*` in `internal/api`). 3. Update documentation if your change affects: - API endpoints → update `docs/src/content/docs/api.md` diff --git a/README.md b/README.md index e9432aac..879abbda 100644 --- a/README.md +++ b/README.md @@ -14,7 +14,7 @@

The open-source real-time API gateway for ClickHouse: schema-aware ingest, async batching, real-time SSE streaming, and tiered query caching. - All in one binary, plus the per-ClickHouse-version artifact it loads at start. + All in one binary, plus the per-ClickHouse-version artifact it loads for your server's line.

@@ -67,7 +67,7 @@ Full walkthrough at **[wavehouse.dev/getting-started](https://wavehouse.dev/gett ## Why WaveHouse? -ClickHouse is a phenomenal OLAP database, but pointing a frontend right at it leaves a lot to be desired: one-row inserts trigger `Too many parts`, there's no backpressure or edge validation, no real-time push, and no row/column security. You end up building custom APIs, a Kafka queue, a batch consumer, a cache tier, and an auth service. **WaveHouse is that whole stack as one binary** (plus a per-ClickHouse-version artifact it loads at start, for ClickHouse-native ingest validation and row-level security) — the only external network dependency is ClickHouse. +ClickHouse is a phenomenal OLAP database, but pointing a frontend right at it leaves a lot to be desired: one-row inserts trigger `Too many parts`, there's no backpressure or edge validation, no real-time push, and no row/column security. You end up building custom APIs, a Kafka queue, a batch consumer, a cache tier, and an auth service. **WaveHouse is that whole stack as one binary** (plus a per-ClickHouse-version artifact it loads for your server's line, for ClickHouse-native ingest validation and row-level security) — the only external network dependency is ClickHouse. If you're building user-facing analytics, WaveHouse is like **Supabase for ClickHouse**. Or an **open-source Tinybird** that pushes data to the frontend in real time over SSE, not just pull-based REST. @@ -121,7 +121,7 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:dev \ Swap in `:vX.Y.Z` and `release.yml` for a release image. Pin the signer either way. `--repo` alone accepts an attestation from any workflow in the repo. -The published images bake the chtypes artifact for ClickHouse 26.8 only. Against any other ClickHouse line, bind-mount a directory holding that line's artifact and set `WH_CHTYPES_REGISTRY` to it (see [chtypes artifacts](https://wavehouse.dev/deployment#chtypes-artifacts)). +The published images bake the chtypes artifact for ClickHouse 26.8 only. Against any other ClickHouse line, bind-mount a directory holding that line's artifact for the container's platform, readable by the image's user, and set `WH_CHTYPES_REGISTRY` to it (see [chtypes artifacts](https://wavehouse.dev/deployment#chtypes-artifacts)). ### C. `go install` (binary, no Docker) @@ -129,13 +129,13 @@ The published images bake the chtypes artifact for ClickHouse 26.8 only. Against go install github.com/Wave-RF/WaveHouse/cmd/wavehouse@latest ``` -`go install` compiles from source with cgo enabled (requires a C toolchain, and on Linux glibc 2.34 or later — Linux amd64/arm64 or macOS arm64) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse loads at start. Fetch it once before the first run: +`go install` compiles from source with cgo enabled (requires a C toolchain, and on Linux glibc 2.34 or later — Linux amd64/arm64 or macOS arm64) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse needs: the server refuses to boot without one. Fetch it once before the first run: ```bash go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch ``` -(From a checkout, `scripts/fetch-chtypes.sh` fetches the build pinned in `chtypes.lock`.) This downloads 160–290 MB into the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision); point `WH_CHTYPES_REGISTRY` elsewhere if you keep it somewhere else. +(From a checkout, `scripts/fetch-chtypes.sh` fetches the build pinned in `chtypes.lock`.) This downloads 160–300 MB into the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision); point `WH_CHTYPES_REGISTRY` elsewhere if you keep it somewhere else. ```bash wavehouse bootstrap ./settings # starter settings directory, every key at its default @@ -158,7 +158,7 @@ You'll need **Go 1.27+, GNU Make 4+, Docker (Compose v2), Node.js 22 LTS, and pn ```bash make tools # one-time bootstrap -scripts/fetch-chtypes.sh # once per machine: the chtypes artifact (160–290 MB) +scripts/fetch-chtypes.sh # once per machine: the chtypes artifact (160–300 MB) docker compose -f deployments/compose/dependencies.yaml up -d clickhouse make dev # hot-reload on .go save ``` diff --git a/SECURITY.md b/SECURITY.md index 694795e7..6813e5bb 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -25,10 +25,10 @@ We will acknowledge receipt within 48 hours and aim to provide an initial assess WaveHouse handles data and enforces strict isolation: - **JWT validation**: The JWT middleware always runs (there is no on/off switch). Signing supports either an HMAC shared secret or a remote JWKS endpoint (`auth.jwks_url`, one verifier per tenant over a nested settings directory, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set — the HMAC secret is boot config shared by every tenant; a JWKS response is capped at 1 MiB). Accepted signing algorithms are restricted to the configured verifier's family — HMAC accepts only `HS256/384/512`, JWKS only the asymmetric set (`RS*`/`ES*`/`PS*`/`EdDSA`) — and the token's `alg` header is validated before any key material is used, so `alg: none` and algorithm-confusion attacks (re-signing with `HS256` against a JWKS deployment's public key) are rejected. A request with no token, or an invalid/expired one, falls back to the policy `default_role`; elevated access requires a valid token, and a denied request that carried a bad token fails loud (`401`) rather than as a bare `403`. With neither a secret nor a JWKS URL configured, no token validates at all: the verifier refuses rather than hand the JWT library an empty HMAC key, which it would accept as a match for a token signed with one. The one token outcome that never falls back is a tenant whose JWKS has not been fetched yet: its token-bearing requests are refused with `503` + `Retry-After` rather than evaluated under a lesser role, so a token is never silently downgraded while the keys that would verify it are still on their way. -- **Role-based access control**: Roles are extracted from a configurable JWT claim path. Non-admin roles have per-table, per-column policies enforced on ingest, query, and the live SSE stream; row-level rules split by path — insert `check` constraints are enforced (and auto-injected) on ingest, while select `filter` predicates apply to structured queries and the live SSE stream (the stream's in-memory row-filter comparison has a documented fail-closed boundary — see the [access-control docs](https://wavehouse.dev/access-control#where-each-rule-is-enforced)); the admin role (`policy.admin_role`, `"admin"` by default, exact case-sensitive match) bypasses them. The configured non-JWT operator key (`auth.operator_key`) likewise bypasses per-role policy — a matching request is authorized as a full-access platform operator without a JWT; treat it as an admin secret. A request presenting a *non-matching* operator key is logged at `WARN` and counted by `wavehouse_auth_operator_key_failures_total`, so probing of that credential is observable and alertable. -- **Input validation**: ingest bodies (JSON, NDJSON, CSV and TSV) are validated against ClickHouse schemas before processing, by ClickHouse's own parser running in-process as a native shared library (the chtypes artifact) loaded by the API process. Every untrusted ingest body therefore reaches native code, which is why a request body is capped at 16 MiB and why the artifact's integrity is pinned (below). +- **Role-based access control**: Roles are extracted from a configurable JWT claim path. Non-admin roles have per-table, per-column policies enforced on ingest, query, and the live SSE stream; row-level rules split by path — insert `check` constraints are enforced (and auto-injected) on ingest, while select `filter` predicates apply to structured queries and the live SSE stream (the stream evaluates the same predicates per subscriber through the in-process ClickHouse parser and withholds any row it cannot definitely admit — see the [access-control docs](https://wavehouse.dev/access-control#where-each-rule-is-enforced)); the admin role (`policy.admin_role`, `"admin"` by default, exact case-sensitive match) bypasses them. The configured non-JWT operator key (`auth.operator_key`) likewise bypasses per-role policy — a matching request is authorized as a full-access platform operator without a JWT; treat it as an admin secret. A request presenting a *non-matching* operator key is logged at `WARN` and counted by `wavehouse_auth_operator_key_failures_total`, so probing of that credential is observable and alertable. +- **Input validation**: ingest bodies (JSON, NDJSON, CSV and TSV) are validated against ClickHouse schemas before processing, by ClickHouse's own parser running in-process as a native shared library (the chtypes artifact) loaded by the API process. Every untrusted ingest body therefore reaches native code, as do the claim-derived constants of insert checks and stream row filters; the request body is capped at 16 MiB, and the artifact's integrity is pinned (below). - **Query passthrough**: Raw SQL via `POST /v1/ops/query` is restricted to the admin role — the same `RequireAdmin` gate as the rest of `/v1/ops/*`. A request with no/invalid token resolves to the `default_role`, which in a production config is not the admin role (setting `default_role` equal to the admin role is a loudly-warned, dev-only escape hatch), so it cannot reach this endpoint — the one exception is a request presenting the configured `auth.operator_key`, which reaches the whole `/v1/ops/*` surface (including this endpoint) without a JWT and even under a deleted policy, so treat that key as an admin secret. Raw SQL has no per-statement scope check (a full SQL parser would be needed to authorize predicates), so the role gate is the entire authorization story. Non-admin callers use structured queries (`POST /v1/query?table={table}`, validated against schema with permission injection) or named pipes (`GET/POST /v1/pipes/{name}`); raw-SQL grants to non-admin roles via the policy engine are no longer supported (the `raw_sql` field on policies has been removed). -- **Supply chain**: The native parser library is pinned in `chtypes.lock` by exact file and sha256 per platform and ClickHouse line, fetched with `scripts/fetch-chtypes.sh --frozen` (which refuses a file whose hash differs), and baked into the published images, so a container makes no download at runtime. Third-party GitHub Actions are pinned to full commit SHAs (enforced by the repository's Actions settings — `sha_pinning_required`). `govulncheck` runs on every push/PR. Dependabot opens weekly grouped PRs for Go modules, GitHub Actions, and the npm packages — one grouped PR covering the docs site, TS SDK, and E2E tests via the root pnpm workspace. Released artifacts ship signed [Sigstore](https://www.sigstore.dev/) build-provenance attestations — verify the container image with `gh attestation verify oci://ghcr.io/wave-rf/wavehouse: --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml` (the rolling `:dev` image is signed by `publish-dev.yml` — swap the `--signer-workflow`; `--repo` alone accepts an attestation from any workflow in the repo), a downloaded release-binary archive with `gh attestation verify --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml`, and the `@wavehouse/sdk` package via its npm provenance badge or `npm audit signatures`. (Provenance covers the published binaries and image, not `go install`, which compiles from source.) +- **Supply chain**: The native parser library is pinned in `chtypes.lock` by exact file and sha256 per platform and ClickHouse line, fetched with `scripts/fetch-chtypes.sh`, which always runs the SDK's fetch with `--frozen --lock chtypes.lock` and so refuses a file whose sha256 differs from the lock, and baked into the published images, so a container makes no download at runtime. Third-party GitHub Actions are pinned to full commit SHAs (enforced by the repository's Actions settings — `sha_pinning_required`). `govulncheck` runs on every push/PR. Dependabot opens weekly grouped PRs for Go modules, GitHub Actions, and the npm packages — one grouped PR covering the docs site, TS SDK, and E2E tests via the root pnpm workspace. Released artifacts ship signed [Sigstore](https://www.sigstore.dev/) build-provenance attestations — verify the container image with `gh attestation verify oci://ghcr.io/wave-rf/wavehouse: --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml` (the rolling `:dev` image is signed by `publish-dev.yml` — swap the `--signer-workflow`; `--repo` alone accepts an attestation from any workflow in the repo), a downloaded release-binary archive with `gh attestation verify --repo Wave-RF/WaveHouse --signer-workflow Wave-RF/WaveHouse/.github/workflows/release.yml`, and the `@wavehouse/sdk` package via its npm provenance badge or `npm audit signatures`. (Provenance covers the published binaries and image, not `go install`, which compiles from source.) ## Disclosure Policy diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 658fbf5f..b6024d89 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -72,7 +72,7 @@ There is no separate `service` role. To reach an admin endpoint or to read data ### Operator key -`auth.operator_key` (config; presented in an `Authorization: Operator ` header, or the `X-Operator-Key` alias) is a static, role-free credential for the person running the deployment, meant for break-glass recovery. A request presenting it is authorized as a **full-access platform operator** — the entire data plane *and* the `/v1/ops/*` management surface — without minting a JWT, and independently of the JWT verifier (`jwt_secret`/`jwks_url`). +`auth.operator_key` (config; presented in an `Authorization: Operator ` header, or the `X-Operator-Key` alias) is a static, role-free credential for the person running the deployment, meant for break-glass recovery. A request presenting it is authorized as a **full-access platform operator** — the `/v1/ops/*` management surface always, and the entire data plane while a policy is adopted (with none, its data-plane requests are denied like anyone else's) — without minting a JWT, and independently of the JWT verifier (`jwt_secret`/`jwks_url`). Unlike `admin_role`, the operator key is honored **even when no policy is adopted** (an empty `policies.json`), so it is the one HTTP credential that still reaches `/v1/ops/*` — `POST /v1/ops/settings/reload` once the file is fixed, whose findings report exactly why a rejected directory was refused — the break-glass case the caution above describes. It is matched in constant time, takes precedence over any Bearer token on the same request, and is disabled when empty (the default). Treat it as an admin secret: load it from a secret store, serve it only over TLS, and rotate it like any other credential. See [Configuration — Authentication](/configuration#authentication). @@ -241,7 +241,7 @@ Any `filter` or `check` operator value may interpolate token claims with `{{ jwt - `{{ jwt.sub }}` → the token's `sub` claim. - `{{ jwt.app_metadata.tenant_id }}` → a nested claim. -Values are always bound as SQL **parameters**, never concatenated into the query, so templating is injection-safe. If a claim path in a `filter` template can't be resolved (a validly-signed token that simply doesn't carry the claim), that filter **fails closed**: the predicate becomes constant-false, so on the structured-query path (`POST /v1/query`) the role sees **no rows**, and the live stream withholds every event for that subscriber — one resolution drives both read surfaces (see [where each rule is enforced](#where-each-rule-is-enforced)). This holds for every operator — `_eq`, `_neq`, `_gt`, `_lt`, and `_in` alike. The alternative, binding the empty string the template would render to, would leave a live predicate against `''`: `_eq` would match every empty-valued row, and `_neq`/`_gt` on a string column would match essentially *all* rows, erasing the restriction. A literal value with no template in it — including an explicit `""` — binds exactly as written, and ClickHouse reads it under the column's type: a `String` column compares that exact text, while a numeric column reads the constant as a number. A spelling the column cannot read is not a second chance — `_eq: "1.0"` against a `UInt64` matches no rows on a read filter and is refused with `403` on an insert (an integer column takes only the canonical spelling; see the caution below), and on another numeric column such as `Decimal` it is ClickHouse's own code 53 `TYPE_MISMATCH` per row, which withholds on a read filter and is a `422` on an insert. Write a literal the column can read. +Values are always bound as SQL **parameters**, never concatenated into the query, so templating is injection-safe. If a claim path in a `filter` template can't be resolved (a validly-signed token that simply doesn't carry the claim), that filter **fails closed**: the predicate becomes constant-false, so on the structured-query path (`POST /v1/query`) the role sees **no rows**, and the live stream withholds every event for that subscriber — one resolution drives both read surfaces (see [where each rule is enforced](#where-each-rule-is-enforced)). This holds for every operator — `_eq`, `_neq`, `_gt`, `_lt`, and `_in` alike. The alternative, binding the empty string the template would render to, would leave a live predicate against `''`: `_eq` would match every empty-valued row, and `_neq`/`_gt` on a string column would match essentially *all* rows, erasing the restriction. A literal value with no template in it — including an explicit `""` — binds exactly as written, and ClickHouse reads it under the column's type: a `String` column compares that exact text, while a numeric column reads the constant as a number. A spelling the column cannot read is not a second chance — `_eq: "1.0"` against a `UInt64` matches no rows on a read filter and is refused with `403` on an insert (an integer column takes only the canonical spelling; see the caution below), and on another numeric column such as `Decimal` it is ClickHouse's own code 53 `TYPE_MISMATCH` (72 on a `Float` column), which fails every `/v1/query` read by that role with `400 clickhouse.rejected`, withholds every row on the stream (reason `error`), and is a `422` on an insert. Write a literal the column can read. "Can't be resolved" means the claim path is **absent** from the token (or `null`) — or resolves to a JSON **object or array** rather than a scalar, which usually means a dropped path segment (`{{ jwt.app_metadata }}` where `{{ jwt.app_metadata.tenant_id }}` was meant); the one structured shape with defined semantics is the bare-claim `_in` array above. Scalar claims — strings, booleans, and numbers — resolve normally. A numeric claim binds in **canonical decimal form**, not the token's spelling: `1.0` and `1e3` bind as `1` and `1000`, so spelling differences between issuers never change the bound value, and an integer id keeps every digit up to a ~100-digit bound on the value's *exact* decimal form (so `1e150` is refused even though a float64 holds it — prefer issuing integer ids as integers). Type the scoped column to match: an integer column, or a `Decimal` whose scale covers the claim's fractional digits, keeps the comparison exact for ids that fit the column's type, while a `Float32`/`Float64` column rounds the stored value and gives that exactness back (`col = '9007199254740993'` matches a stored `9007199254740992` there). A claim that is *present but empty* is a value the token vouches for: it resolves to `''` and binds normally. Make sure your identity provider issues the claims your policy templates reference — and omits unset claims rather than issuing them as empty strings. @@ -252,7 +252,7 @@ Compare claims against `String` or `UUID` columns (tenant ids, org ids, user ids On an integer column (`UInt8` through `UInt256`, `Int8` through `Int256`, `Nullable` included) the claim must still fit the column's type, and is compared through a strict cast: a claim that is not the canonical spelling of a value the column can hold — out of range such as `18446744073709551616` (2^64), or spelled `007` or `+5` — matches no rows on any operator, and an insert `check` refuses the record with `403`. It never wraps onto another value. -On a timestamp column (`Date`, `DateTime`, `DateTime64`) the claim is compared the way ClickHouse compares a string with that type. Write it as `YYYY-MM-DD hh:mm:ss[.fff]` with no offset: it is read in the column's time zone, or the server's when the column declares none. On the read path, where the claim is bound as a plain `String`, a claim in RFC 3339 form such as `2026-01-01T00:00:00Z` is refused on a `DateTime` column (ClickHouse code 53, measured on 24.8 and 26.8) rather than matched. An insert `check` literal is parsed under `best_effort`, so an offset spelling such as `2026-06-21T06:00:00+02:00` is accepted there and stored as its instant. For a `filter`, use the zone-less form, or compare claims against a String or UUID column instead. +On a timestamp column (`Date`, `DateTime`, `DateTime64`) the claim is compared the way ClickHouse compares a string with that type. On a `Date` column write it as `YYYY-MM-DD`; a time part is refused (code 53). On a `DateTime`/`DateTime64` column write it as `YYYY-MM-DD hh:mm:ss[.fff]` with no offset: it is read in the column's time zone, or the server's when the column declares none. ClickHouse 26.8 reads that comparison under its default `cast_string_to_date_time_mode=best_effort` (the default since 26.5), so an RFC 3339 claim such as `2026-01-01T00:00:00Z`, or one with an offset, matches the instant it names on `/v1/query` and on the stream alike (measured on 26.8.15.10). On a line before 26.5 (24.8, 25.8), where `basic` is still the default, the RFC 3339 spelling is refused: on `/v1/query` with code 53, failing every read by that role with `400 clickhouse.rejected`, and on the stream, whose parser follows its line's default, by withholding every row (reason `error`; measured on the 25.8 artifact). A 26.8 server whose profile sets `basic` refuses it on `/v1/query` while the stream still matches it. An insert `check` literal is parsed under `best_effort`, so an offset spelling such as `2026-06-21T06:00:00+02:00` is accepted there and stored as its instant. For a `filter`, the zone-less form is the one spelling every server reads the same way; or compare claims against a String or UUID column instead. Prefer identity columns over time columns, and express time windows in the query instead. ::: @@ -288,9 +288,9 @@ Checks are compiled into **one chtypes filter** — `col = {p:String}` for `_eq` - **If the request body includes the column**, its value must satisfy the check or the record is rejected with `403 check failed for column "x"` (`… for columns "x", "y"` when more than one is checked — the filter is AND-joined, so it names the set tested rather than inventing an attribution). Comparison is ClickHouse's, under the column's own type: on an integer, `Decimal`, `UUID` or `String` column a writer cannot forge a row for another tenant, because equal values store equal. A `Float32`/`Float64` column rounds the stored value and gives that exactness back — the row can land on a neighboring id. A check the engine cannot evaluate at all is a `422`, never a silent pass. - **If the request body omits the column** — or sends an explicit `null` on a non-`Nullable` column, which `input_format_null_as_default` resolves the same way — an `_eq` check **auto-injects** the claim-derived value, so clients can send just the business fields and let the policy stamp `user_id` and `tenant_id` from the token. On a `Nullable` column an explicit `null` is stored as `NULL`, which fails the check (`403`). It is implemented as a `DEFAULT` on the role's compiled schema, so a value the caller *does* send still wins. An `_eq` check also covers a column the role may not otherwise write (left out of `allow_columns`, or in `deny_columns`): a record that omits it is filled with the required value, one that supplies exactly that value is accepted, and any other value fails the check (`403`). An `_in` check has no single value to stamp, so the **table's own default** is what gets tested: the record is admitted if that default is in the claim-derived set and rejected if it is not. - **The check's column must be one a record can actually carry.** A `check` naming a column the table does not have, one ClickHouse computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is refused with a per-record `403` naming the column — on *every* insert by that role, until the policy or the table is corrected. None of the three can be enforced: the published row has one slot per wire column, so an injected value for a computed or unknown column is dropped on the way out, and an ephemeral column is never stored. Each would have answered `200` while enforcing nothing. `wavehouse validate` cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**. -- **A claim literal the column cannot read fails closed, not loosely.** A `_eq` value that is not a legal literal for the column (`"1.0"` on a `UInt64`) cannot be compiled as that column's `DEFAULT`, so the injection is dropped and logged; the filter then judges the record as sent, which an absent column loses. On an integer column every record then fails the check — a `403`; on another column a supplied value is ClickHouse's code 53 `TYPE_MISMATCH` per row — a `422`. +- **A claim literal the column cannot read fails closed, not loosely.** A `_eq` value that is not a legal literal for the column (`"1.0"` on a `UInt64`) cannot be compiled as that column's `DEFAULT`, so the injection is dropped and logged; the filter then judges the record as sent, which an absent column loses. On an integer column every record then fails the check — a `403`; on another column the comparison itself errors on every record, supplied or omitted (ClickHouse's code 53 `TYPE_MISMATCH` on a `Decimal`, 72 on a `Float`) — a `422`. -**When the claim can't be resolved** (a validly-signed token that doesn't carry it), an `_eq` check is — unlike a row filter — **not** fail-closed on a `String` column: the template still renders, the unresolvable placeholder replaced by the empty string and any surrounding literal text kept (`"acct-{{ jwt.org_id }}"` → `acct-`, a bare `"{{ jwt.sub }}"` → `''`), and that rendered value becomes the required value — so an omitted column is auto-injected with it and any other supplied value is rejected. On a **numeric** column it now fails closed: `''` is not a literal a `UInt64` can read, so the role's schema will not compile with it as a default and no record satisfies the check — every insert by that role into that table is refused (a `403` on an integer column, a `422` on a `Decimal` or `Float` one), rather than the rows landing on tenant/user `0` as they used to ([#463](https://github.com/Wave-RF/WaveHouse/issues/463)). An unresolvable `_in` check fails closed everywhere: the claim-derived set is empty, so nothing satisfies it. The `_in` element rule from [row-level security](#row-level-security) applies here too — one non-scalar element (`null`, object, nested array) in the claim array collapses the allowed set the same way. +**When the claim can't be resolved** (a validly-signed token that doesn't carry it), an `_eq` check is — unlike a row filter — **not** fail-closed on a `String` column: the template still renders, the unresolvable placeholder replaced by the empty string and any surrounding literal text kept (`"acct-{{ jwt.org_id }}"` → `acct-`, a bare `"{{ jwt.sub }}"` → `''`), and that rendered value becomes the required value — so an omitted column is auto-injected with it and any other supplied value is rejected. On a column whose type cannot read `''` — a number, a `Date`/`DateTime`, a `UUID` — it now fails closed: `''` is not a literal a `UInt64` can read, so the role's schema will not compile with it as a default and no record satisfies the check — every insert by that role into that table is refused (a `403` on an integer column, a `422` on any other), rather than the rows landing on tenant/user `0` as they used to ([#463](https://github.com/Wave-RF/WaveHouse/issues/463)). An unresolvable `_in` check fails closed everywhere: the claim-derived set is empty, so nothing satisfies it. The `_in` element rule from [row-level security](#row-level-security) applies here too — one non-scalar element (`null`, object, nested array) in the claim array collapses the allowed set the same way. This pairs naturally with a matching `filter` on the `select` side: `check` stamps the tenant on write, `filter` scopes reads to that tenant. @@ -324,7 +324,7 @@ Four fields cap the cost of a single structured query for this role. All must be | Field | Effect | | ----- | ------ | -| `max_rows` | Caps the query's `LIMIT`. If the caller asks for more (or omits a limit), the result is clamped to `max_rows`. | +| `max_rows` | Caps the query's `LIMIT`. It can only lower the tenant's `query.default_max_rows` ceiling, never raise it: a caller asking for more, or omitting a limit, gets the smaller of the two. | | `max_execution_time` | Caps the query execution time, applied as the *minimum* of this value and the server's `clickhouse.query_timeout`. | | `max_rows_to_read` | Caps the rows **scanned from storage** server-side — the lever that stops a full-table scan. The read is rejected once it is exceeded. | | `max_memory_usage` | Caps **peak query memory** server-side — the lever that stops a heavy aggregation from exhausting the box. The read is rejected once it is exceeded. | @@ -365,9 +365,9 @@ The same policy drives every data path, but not every field is meaningful on eve | Named pipe | `GET/POST /v1/pipes/{name}` | per-pipe `allowed_roles` (not the policy engine; see [Named Pipes](/pipes)). Resource limits come from ClickHouse's [server-wide settings](/configuration#server-side-resource-limits), not per-role policy caps | :::caution[Live streams enforce column and row policy, but not resource limits] -SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. +SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause on a server running that line's default settings, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, ClickHouse's code 53; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine cannot compile (a constant the column type cannot read) withholds every row for that role until the filter or the table changes — logged once, not per event. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72), or the parse of the event failed outright; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. @@ -397,7 +397,7 @@ The reload endpoint's request and response shapes live in the [API Reference](/a ## The policy lifecycle -The settings directory is the **source of truth** for the live policy — there is no other copy: +The settings directory is the **source of truth** for the live policy — there is no other copy. Over a flat directory (a [nested one](/deployment#the-nested-settings-directory) fails closed per tenant instead): - At boot, `policies.json` (with `roles.json`, `pipes.json`, and `config.json`) must exist, parse, and pass validation, or WaveHouse refuses to start. That turns a typo, a missing mount, or a policy invalid under a newly-tightened rule — such as the [malformed-template rejection](#jwt-claim-templating) above — into a loud failure instead of a silent fail-closed deployment that denies everything. - An **empty** `policies.json` (`{}`) is valid but adopts no policy — every **token-based** request is denied (logged loudly, admin included). `wavehouse bootstrap` writes it that way, so a fresh directory is closed until you write a policy. @@ -462,10 +462,10 @@ Per-role permissions (`tables.
..select` and `.insert`): | `allow_columns` | string[] | select, insert | Allowlist of columns. Empty or `["*"]` = all columns (minus `deny_columns`). | | `deny_columns` | string[] | select, insert | Blocklist of columns. Always wins over `allow_columns`. | | `filter` | map | select | Row-level predicates (`_eq`/`_neq`/`_gt`/`_lt`/`_in`), ANDed together — injected as a SQL `WHERE` on structured reads and evaluated per subscriber on the live stream (see [enforcement](#where-each-rule-is-enforced)). `_in` takes a single claim → `col IN (…)`: an array-valued claim contributes each element, a scalar claim acts as a one-element set. An entry naming no operator (`{}`) is rejected when the policy is validated. Values support `{{ jwt.path }}` templating; a claim the token doesn't carry fails the filter closed (no rows on either surface), and a malformed template is rejected when the policy is validated. | -| `check` | map | insert | Required insert values (`_eq`, or `_in` for a claim-derived set; `_neq`/`_gt`/`_lt`, setting both `_eq` and `_in`, and an entry naming no operator at all, are rejected when the policy is validated). `_eq` is enforced if present and auto-injected if absent; `_in` requires the column be present and in-set. Supports templating. | +| `check` | map | insert | Required insert values (`_eq`, or `_in` for a claim-derived set; `_neq`/`_gt`/`_lt`, setting both `_eq` and `_in`, and an entry naming no operator at all, are rejected when the policy is validated). `_eq` is enforced if present and auto-injected if absent; `_in` tests the stored value: a supplied value must be in the set, and an omitted column is judged on the table's own default. Supports templating. | | `allowed_aggregations` | string[] | select | Allowlist of aggregation functions. Empty = all (minus denied). Case-insensitive. | | `denied_aggregations` | string[] | select | Blocklist of aggregation functions. Always wins. | -| `max_rows` | int | select | Caps the query `LIMIT`. `0` = no limit (the `query.default_max_rows` config default applies). Must be non-negative. | +| `max_rows` | int | select | Caps the query `LIMIT`, below the tenant's `query.default_max_rows` ceiling (it never raises it). `0` = no role cap (`query.default_max_rows` alone applies). Must be non-negative. | | `max_execution_time` | duration or ms | select | Caps the query timeout (min with `clickhouse.query_timeout`). Set as `"5s"`/`"500ms"` or a number of ms. `0` = no limit. Must be non-negative. | | `max_rows_to_read` | int | select | Caps rows **scanned** server-side (ClickHouse `max_rows_to_read`); the read is rejected once exceeded. `0` = no role limit. Must be non-negative. | | `max_memory_usage` | size or bytes | select | Caps peak query memory server-side (ClickHouse `max_memory_usage`). Set as `"4GiB"`/`"512MiB"` or a number of bytes. `0` = no role limit. Must be non-negative. | diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 23200082..afb6753d 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -79,7 +79,7 @@ For SSE, streaming endpoints, or any handler that has already started writing th ### ClickHouse errors on the query paths -When ClickHouse fails a query on [`POST /v1/query`](#post-v1querytabletable--structured-query), [`/v1/pipes/{name}`](#getpost-v1pipesname--execute-named-pipe) or [`POST /v1/ops/query`](#post-v1opsquery--query-clickhouse), the status comes from **what kind of failure it was**, not from ClickHouse's HTTP status: ClickHouse answers a syntax error, a missing grant and an overloaded server alike with HTTP `500`. WaveHouse reads the ClickHouse exception code (the `X-ClickHouse-Exception-Code` header or the `Code: NNN.` in the message) and answers with two extra fields alongside `error`: +When ClickHouse fails a query on [`POST /v1/query`](#post-v1querytabletable--structured-query), [`/v1/pipes/{name}`](#getpost-v1pipesname--execute-named-pipe) or [`POST /v1/ops/query`](#post-v1opsquery--query-clickhouse), the status comes from **what kind of failure it was**, not from ClickHouse's HTTP status, which does not say which kind of failure it was: on 26.8 a syntax error is `400`, an unknown table `404` and a `TIMEOUT_EXCEEDED` `408` — the same statuses a proxy in front uses to mean something else. WaveHouse reads the ClickHouse exception code (the `X-ClickHouse-Exception-Code` header or the `Code: NNN.` in the message) and answers with two extra fields alongside `error`: ```json {"error": "Code: 62. DB::Exception: Syntax error: …", "code": "clickhouse.rejected", "retryable": false} @@ -92,7 +92,7 @@ When ClickHouse fails a query on [`POST /v1/query`](#post-v1querytabletable--str | 403 | `clickhouse.access_denied` | `false` | The ClickHouse user WaveHouse connects as lacks a grant the statement needs (`ACCESS_DENIED`). Grant it, or run something it may. Logged at `WARN` too | | 502 | `clickhouse.misconfigured` | `false` | ClickHouse refused the credentials or database WaveHouse connects with: a wrong password, an unknown or expired user, a refused address, the database denied, a `401`/`403` from a proxy in front of it, or any other `3xx`/`4xx` with no exception code — a wrong path, a redirect WaveHouse does not follow, or a read the server refused as a write (`READONLY`, from a `readonly=1` profile or a statement the mutation classifier missed) (a codeless `408`/`429` is `503`, a `413` is `400 clickhouse.rejected`). Every query fails until the operator fixes the tenant's `clickhouse` settings or `WH_CH_PASSWORD`, so retrying does not help. Logged at `WARN` | | 502 | `clickhouse.response_too_large` | `false` | The response outgrew the 64 MiB the reader buffers (`/v1/query`, pipes and `/v1/ops/query` alike). Narrow the query or add a `LIMIT` | -| 503 | `clickhouse.unavailable` | `true` | ClickHouse, or the way to it, could not take the query now: connection refused or dropped, a timeout, too many queries, memory pressure, lost replicas or Keeper, or a `502`/`503`/`504`/`429`/`408` from a proxy. `Retry-After: 5` | +| 503 | `clickhouse.unavailable` | `true` | ClickHouse, or the way to it, could not take the query now: connection refused or dropped, a timeout, too many queries, memory pressure, lost replicas or Keeper, or a `502`/`503`/`504`/`429`/`408` from a proxy. `Retry-After: 5`. `READONLY` lands here on `/v1/ops/query` and write pipes, even from a permanent `readonly=1` profile — only a read sent under `readonly=2` reads it as the `502` above | | 500 (`/v1/query`, pipes) / 502 (`/v1/ops/query`) | `clickhouse.unknown` | `true` | A failure with no verdict: no exception code and no recognizable transport error | A [pipe that writes](/pipes#pipes-that-write) answers with the same status and `code`, but always `retryable: false` and with no `Retry-After`: the statement may have run, so a retry could run it twice. @@ -128,7 +128,7 @@ Status code: `503 Service Unavailable` The boot-degraded response lets an operator `curl /livez` to learn why the gateway isn't ready to serve traffic yet, instead of grepping a restart-loop log. The binary is bound on `:8080` and serves diagnostics, but is not yet accepting ingest/query traffic. Schema discovery retries with jittered exponential backoff (each wait a random time below a bound that doubles from 2s to 60s); once a Refresh succeeds, `/livez` flips to `200` and stays there for the rest of the process lifetime — transient ClickHouse blips after that point are reflected in `/readyz`, not `/livez`. -Over a [nested settings directory](/deployment#the-nested-settings-directory) the probe reads every tenant together: `/livez` is `503` while **no** tenant has completed a first discovery — the diagnostic names the tenant whose attempt it reports (`schema discovery: tenant acme: …`), and reads `no tenant has completed a first discovery yet` before any attempt, when the directory serves no tenant, and once the tenant it named stops being served — and `200` from the first tenant's success on, for the rest of the process lifetime. A tenant whose ClickHouse is unreachable after that is a log line and the `wavehouse_schema_refresh_failures_total{tenant}` counter, never a probe failure. A tenant that has not completed its own first discovery answers `503` (`schema not loaded yet`) on its schema-aware routes until it does; one whose ClickHouse goes down after that answers query errors, as a single-tenant server does. +Over a [nested settings directory](/deployment#the-nested-settings-directory) the probe reads every tenant together: `/livez` is `503` while **no** tenant has completed a first discovery — the diagnostic names the tenant whose attempt it reports (`schema discovery: tenant acme: …`), and reads `schema discovery: no tenant has completed a first discovery yet` before any attempt, when the directory serves no tenant, and once the tenant it named stops being served — and `200` from the first tenant's success on, for the rest of the process lifetime. A tenant whose ClickHouse is unreachable after that is a log line and the `wavehouse_schema_refresh_failures_total{tenant}` counter, never a probe failure. A tenant that has not completed its own first discovery answers `503` (`schema not loaded yet`) on its schema-aware routes until it does; one whose ClickHouse goes down after that answers query errors, as a single-tenant server does. --- @@ -223,7 +223,7 @@ Every other route answers `404`, including every tenant route. Under `/v1/ops`, Validates a body of records against the ClickHouse schema for `{table}` and publishes each accepted one to the message queue. Returns immediately — ClickHouse insertion happens asynchronously via the batch consumer. A single-object body answers `{"ok":true}` (or `{"duplicate":true}` when dedup is on); every other body answers the [batch summary](#batch-ingest). -**The body goes to ClickHouse's own parser as-is.** WaveHouse never decodes a record: validation, type coercion, `DEFAULT` substitution and timestamp parsing are ClickHouse's own, running in-process via [chtypes](/deployment#chtypes-artifacts) (`internal/typelayer`) — the exact code path a real `INSERT` runs. A rejection therefore carries ClickHouse's own message and its numeric error number as `exception_code` rather than a WaveHouse-authored sentence, and there is no separate coercion table to keep in sync with the server. A rejected single record answers `{"error","exception_code"}` with no string `code`; only a refusal of the whole request (a `header=present` header naming a column the table lacks) carries both, `code: "clickhouse.rejected"` and `exception_code`. The SDK reports a rejected single record as `HTTP_400`, with the number in `error.details`. +**The body goes to ClickHouse's own parser as-is.** WaveHouse never decodes a record: validation, type coercion, `DEFAULT` substitution and timestamp parsing are ClickHouse's own, running in-process via [chtypes](/deployment#chtypes-artifacts) (`internal/typelayer`) — the same parser a real `INSERT` runs, under the settings WaveHouse pins. A rejection therefore carries ClickHouse's own message and its numeric error number as `exception_code` rather than a WaveHouse-authored sentence, and there is no separate coercion table to keep in sync with the server. A rejected single record answers `{"error","exception_code"}` with no string `code`; only a refusal of the whole request (a `header=present` header naming a column the table lacks) carries both, `code: "clickhouse.rejected"` and `exception_code`. The SDK reports a rejected single record as `HTTP_400`, with the number in `error.details`. **`Content-Type` is required and authoritative**: it declares the format and the bytes never override it. @@ -236,7 +236,7 @@ Validates a body of records against the ClickHouse schema for `{table}` and publ | `text/csv; header=present`, `text/tab-separated-values; header=present` | a header line naming the columns, in any order — see [Header formats](#header-formats-headerpresent) | | anything else, or none | `415`, listing the accepted types | -The two JSON families are one format to ClickHouse; the declaration decides only how the body frames its records. The single thing the body still chooses is *arity within `application/json`*: the first non-whitespace byte picks an array (`[`) or a single object. Under a single-object body only the first object is read — concatenated objects after it are ignored, a `200` for one record; declare NDJSON for anything line-framed ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). The reverse now works: a JSON array declared `application/x-ndjson` ingests every element. +The two JSON families are one format to ClickHouse; the declaration decides only how the body frames its records. The single thing the body still chooses is *arity within `application/json`*: the first non-whitespace byte picks an array (`[`) or a single object. Under a single-object body only the first object is answered — concatenated objects after it are parsed but neither published nor reported, a `200` for one record (one cut off mid-record can still turn that answer into a `422` decline); declare NDJSON for anything line-framed ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). The reverse now works when every element is valid: a JSON array declared `application/x-ndjson` ingests every element. It is not re-framed, though, so one bad element in a single-line array makes chtypes decline the whole body (`422` for every record); declare `application/json` to get per-record answers. :::note[What counts as a valid declaration] The header is parsed with Go's `mime.ParseMediaType` (RFC 9110 §8.3) and the **media type** decides the format, so no malformed *parameter* costs the request — `application/json; charset`, `application/json;;`, a value left mid-quote, a name repeated with different values all read as `application/json`. The one parameter that also decides a format is `header`, on `text/csv` and `text/tab-separated-values` only: `present` selects the header format, `absent` the strictly positional one, no `header` at all ClickHouse's default reading, and any other value is a `415`. A line whose parameters did not parse and that mentions `header` is a `415` as well, because guessing at it could ingest a declared header line as data or drop a data row as a header. Two more things are refused. A malformed parameter on a line that **also contains a comma** is a `415`, because the comma may be a second declaration joined on and the error cannot tell that from a comma inside data ([#563](https://github.com/Wave-RF/WaveHouse/issues/563)) — so `application/json; profile="a,b"` is fine and `application/json; profile="a,b"; charset` is not. And `Content-Type` is a **singleton** field (§5.3 forbids repeating it), so repeated header *lines* are accepted only when they agree, while a comma-joined value is refused outright: §8.3 warns that picking a member of the resulting pseudo-list is itself an interoperability and security hazard. @@ -254,20 +254,21 @@ The policy engine authorizes mutations by inspecting the columns being written. **What ClickHouse decides, and what WaveHouse decides.** Everything about a *value* is ClickHouse's: -- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x`. So are `MATERIALIZED` and `ALIAS` columns — neither is ever part of a published row. (An `_eq` [insert check](/access-control#insert-checks) on a column the role may not write is the one exception: it accepts exactly the required value, and the row carries it.) An `EPHEMERAL` column is accepted as input only where the format names its columns (the JSON family and the `…WithNames` formats), the role may write it, a `DEFAULT` column reads it, and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column reads it; its value feeds that `DEFAULT` and is never stored, selected or published. Any other `EPHEMERAL` column is refused like an unknown one (code **117**): ClickHouse computes `MATERIALIZED` columns at insert time from the published row, which never carries the ephemeral value, so accepting it there would drop it silently. A positional CSV or TSV body (no header, or `header=absent`) carries the wire columns only. +- A field the role may not write is indistinguishable from one the table does not have: both are code **117**, `Unknown field found while parsing JSONEachRow format: x` (WaveHouse pins `input_format_skip_unknown_fields=0`, so an unknown field is refused where a default 26.8 `INSERT` would silently drop it). So are `MATERIALIZED` and `ALIAS` columns — neither is ever part of a published row. A field name matches its column case-insensitively, as on ClickHouse 26.8 (`{"PAGE": …}` fills `page`). (An `_eq` [insert check](/access-control#insert-checks) on a column the role may not write is the one exception: it accepts exactly the required value, and the row carries it.) An `EPHEMERAL` column is accepted as input only where the format names its columns (the JSON family and the `…WithNames` formats), the role may write it, a `DEFAULT` column reads it, and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column reads it; its value feeds that `DEFAULT` and is never stored, selected or published. Any other `EPHEMERAL` column is refused like an unknown one (code **117**): ClickHouse computes `MATERIALIZED` columns at insert time from the published row, which never carries the ephemeral value, so accepting it there would drop it silently. A positional CSV or TSV body (no header, or `header=absent`) carries the wire columns only. - An omitted column takes its `DEFAULT` expression — evaluated by ClickHouse, including a volatile one like `now()` — or the type's default where none is declared (`NULL` on a `Nullable` column), exactly as an `INSERT` naming fewer columns does. An explicit `null` does the same on a non-`Nullable` column (WaveHouse pins `input_format_null_as_default`); on a `Nullable` column it stores `NULL`. -- A coercion ClickHouse would make it makes here (a numeric string into an `Int*`, `"true"` into a `Bool`, an out-of-range integer wrapping); anything it would refuse fails synchronously in the ingest response with its real code, rather than surfacing later in the DLQ. `Nullable()` and `LowCardinality()` wrappers are transparent. +- A coercion ClickHouse would make it makes here (a numeric string into an `Int*`, JSON `true` into an `Int*` as `1`, an out-of-range integer wrapping; a quoted `"true"` into a `Bool` it refuses, code 467); anything it would refuse fails synchronously in the ingest response with its real code, rather than surfacing later in the DLQ. `Nullable()` and `LowCardinality()` wrappers are transparent. -WaveHouse decides only policy: whether the role may insert at all, and whether the record satisfies the role's [`check` clauses](/access-control#insert-checks) — evaluated by the same compiled-filter engine as row-level security, in the same parse that validates the record and against the row ClickHouse produced, so a check sees stored values rather than the payload's spelling. A record ClickHouse refuses reports that refusal, never a check result. A record chtypes cannot evaluate at all — as opposed to accepting or rejecting it — is **declined** (`422`), which is not a data verdict. +WaveHouse decides only policy: whether the role may insert at all, and whether the record satisfies the role's [`check` clauses](/access-control#insert-checks) — evaluated by the same compiled-filter engine as row-level security, in the same parse that validates the record and against the row ClickHouse produced, so a check sees stored values rather than the payload's spelling. A record ClickHouse refuses reports that refusal, never a check result. A record chtypes did not answer for — as opposed to accepting or rejecting it — is **declined** (`422`), which is not a verdict on that record's data: it can be the shape, or a body the reader could not get through as a whole (below), in which case every record in the body is declined. **Error responses.** Rows marked **per-record** are reported in `results` on a batch body (the request itself stays `200`) and become the response status on a single-object body; every other row fails the whole request. | Status | Body | Cause | | ------ | ---- | ----- | -| 400 | `{"error":"","exception_code":}` | **Per-record.** ClickHouse's parser refused the record; `exception_code` and the message are its own. `117` is an unknown field — which now includes a column the role may not write and any `MATERIALIZED`/`ALIAS` column; `27`/`26` are unparseable input; `6` out of range | +| 400 | `{"error":"","exception_code":}` | **Per-record.** ClickHouse's parser refused the record; `exception_code` and the message are its own. `117` is an unknown field — which now includes a column the role may not write and any `MATERIALIZED`/`ALIAS` column; `27`/`26` are input it cannot parse and `33` a record cut off mid-object; other codes are type-specific (`41` a `DateTime`/`DateTime64`, `38` a `Date`, `69` a `Decimal` with too many digits, `72` a number it cannot read — no digits, or a negative into a `UInt*`, `376` a `UUID`, `467` a `Bool`, `691` an unknown `Enum` element, `675` an `IPv4`). An out-of-range integer is not rejected: it wraps (above) | | 400 | `{"error":"Unknown field found in format header: 'x' at position 1 …","code":"clickhouse.rejected","exception_code":117}` | A `header=present` body whose header names a column the table — or the role's writable set — does not have, or names one twice. ClickHouse refuses the body before reading any record, so the whole request fails and nothing is published | -| 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26 or 27) | -| 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | +| 400 | `{"error":"missing table"}` | No `table` query parameter. Checked before the body is read | +| 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26, 27 or 33) — or, cut right after a key's `:` or an opening quote, a `422` decline of every record in the body | +| 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes, or only whitespace (only the first 512 bytes are looked at, so a body opening with 512 bytes of whitespace counts as empty too). A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | | 400 | `{"error":"invalid json: content after the closing ']' of the json array"}` | A body declared `application/json` opening with `[` has something other than whitespace after its closing `]`. That tail is not a record of the array, so the whole request fails and nothing is published | | 400 | `{"error":"missing dedupe id field \"event_id\""}` | **Per-record.** Only with `dedupe.require_id: true`, when the row carries no value for the configured `id_field` (an absent column, a `null` cell or an empty string — the value an omitted `String` id column stores). With `require_id: false` (the default) the row is published un-deduped instead. Either way it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` | @@ -279,13 +280,13 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 404 | `{"error":"unknown table: ..."}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | -| 422 | `{"error":"validation engine declined: "}` | **Per-record.** chtypes could not evaluate the record at all — the artifact declined the shape, rather than the data being wrong. A `check` clause that could not be evaluated lands here too (`validation engine declined: the insert check for column "x" could not be evaluated`) | +| 422 | `{"error":"validation engine declined: "}` | **Per-record.** chtypes did not answer for the record. That can be the shape (the artifact declined it), or a body it could not read as a whole — a record cut off right after a key's `:` or an opening quote, or a single-line array declared NDJSON holding a bad element — in which case every record in the body, good ones included, gets this answer and nothing is published. A `check` clause that could not be evaluated lands here too (`validation engine declined: the insert check for column "x" could not be evaluated`) | | 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither the record's fault nor an unavailable tenant; logged. Nothing was published | | 500 | `{"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` | The role's insert permissions do not compile against the table (for example, the role may write no column of it). It persists until the policy or the table changes, so it carries no `Retry-After` and must not be retried; the cause is in the server log | | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now: it is not open (for example, it failed to open on a reload), or a DynamoDB table is throttling, timing out or unreachable; `Retry-After: 5`. Nothing was published, so the retry is safe | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet (its ClickHouse unreachable, or [no pool for it](/settings-directory#clickhouse)), so whether the table exists is not known; `Retry-After: 5`. Decided before the body is read | -| 500 | `{"error":"publish failed"}` | Message queue error whose outcome is unknown, other than a full queue or an unreachable broker (below): the event may have been stored. With dedupe on, the record's id is left to lapse with the dedupe lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default) rather than given back: a retry inside the lease answers the in-flight `503`, and one after it is published under the same idempotency key, which the queue drops if the first copy was stored. The queue's duplicate window (two minutes) covers up to ~2×lease plus a margin, not just the lease itself, so a retry timed off `Retry-After` anywhere in this flow stores no second copy; a much later one is stored again. | +| 500 | `{"error":"publish failed"}` | Message queue error whose outcome is unknown, other than a full queue or an unreachable broker (below): the event may have been stored. With dedupe on, the record's id is left to lapse with the dedupe lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default) rather than given back: a retry inside the lease answers the in-flight `503`, and one after it is published under the same idempotency key, which the queue drops if the first copy was stored. The queue's duplicate window (two minutes on the embedded broker; under [`mq.backend: nats`](/deployment#external-nats), the partition stream's own `duplicate_window`, which WaveHouse requires to cover the same span) covers up to ~2×lease plus a margin, not just the lease itself, so a retry timed off `Retry-After` anywhere in this flow stores no second copy; a much later one is stored again. | | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure, for that tenant alone) or not open (see [Message Queue](/settings-directory#message-queue)). Under [`mq.backend: nats`](/deployment#external-nats), the table's partition stream is full, which refuses every table in it, or the tenant's table holds as many unwritten rows as the stream allows one subject. Response includes `Retry-After: 30` header. With dedupe on, the record's id is given back, so the retry is published rather than reported as a duplicate. | | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`, a transient broker failure, not a full queue). Only under [`mq.backend: nats`](/deployment#external-nats), including a partition stream the operator deleted; the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the record's id is left to lapse rather than given back, so a retry cannot land as a second copy; `Retry-After` is that lease, rounded up to whole seconds, when dedupe was on for the record, else the flat `Retry-After: 5`. | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | Dedupe is on and another request carrying the same id is still being published — usually a client's timeout-retry racing its own original. Its outcome decides whether this record is a duplicate, so retry after the `Retry-After` header (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). | @@ -303,13 +304,13 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ #### Timestamp rendering -WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling ClickHouse's own parser accepts under `date_time_input_format=best_effort` (the setting WaveHouse pins, both at chtypes' ingest compile and on the worker's `INSERT`) is accepted — RFC 3339 with any offset, a zone-less `YYYY-MM-DD[ T]HH:MM:SS[.fff]` read in the column's declared zone else the server's default, a Unix-seconds string, a bare integer (ClickHouse 26.8 reads a bare number in a `DateTime64` column as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — send milliseconds as a quoted string, or as a decimal number of seconds), among the other forms its lenient parser reads. It is ClickHouse's grammar, not a reimplementation of it, so whatever a real `INSERT` into this table would accept, ingest accepts, with the same coercions and the same refusals. +WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling ClickHouse's own parser accepts under `date_time_input_format=best_effort` (the setting WaveHouse pins, both at chtypes' ingest compile and on the worker's `INSERT`) is accepted — RFC 3339 with any offset, a zone-less `YYYY-MM-DD[ T]HH:MM:SS[.fff]` read in the column's declared zone else the server's default, a Unix-seconds string, a bare integer (ClickHouse 26.8 reads a bare number in a `DateTime64` column as epoch **seconds**, so an epoch-millisecond number clamps to `9999-12-31` — send milliseconds as a quoted string, or as a decimal number of seconds), among the other forms its lenient parser reads. It is ClickHouse's grammar, not a reimplementation of it, so a value a real `INSERT` into this table would accept under the settings WaveHouse pins, ingest accepts, with the same coercions and the same refusals. **Outbound**, `DateTime`/`DateTime64` values in the NATS/SSE wire row and in `/v1/query` / `/v1/pipes/{name}` results are the exact bytes ClickHouse's writer produces, as RFC 3339 in UTC, whatever zone the column declares (ClickHouse's `date_time_output_format=iso`): `"2026-06-21T04:00:00.123Z"` for a `DateTime64(3)`, `"2026-06-21T04:00:00Z"` for a `DateTime`, with the fraction at the column's own precision, trailing zeros kept (a `DateTime64(3)` on a whole second is `"…:00.000Z"`; releases before this one trimmed it to `"…:00Z"`). An event published before an upgrade to this spelling, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in. Every consumer renders from the same stored value the same way, so SSE, `/v1/query` and `/v1/pipes/{name}` agree on spelling for a given row **by construction**, with no WaveHouse rewriting step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)), and the strings order the same way the instants do. The raw-SQL proxy `/v1/ops/query` sets the same `date_time_output_format=iso`, so it spells a `DateTime`/`DateTime64` the same way; its other types are returned as ClickHouse renders them under `FORMAT JSON`. `Date`, `Date32` and `Time` columns keep their own form (`"2026-06-21"`) throughout. **One server time zone per ClickHouse line, per process.** The in-process parser takes its zone once, when a process first opens a ClickHouse line, and keeps it for the process's lifetime. A tenant whose server reports a different zone than the one that line was opened with is refused on its own — ingest answers `503`, a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working. Run tenants whose servers use different zones in separate processes. -**Row-level security compares instants, not spellings.** A stream row filter on a `DateTime`/`DateTime64` column is compiled and evaluated by the same engine that validates ingest ([internal/typelayer](/access-control#where-each-rule-is-enforced)), so a filter constant in any spelling ClickHouse would accept in a `WHERE` clause matches the stored instant however the payload spelled it, and a predicate the engine cannot compile withholds every row for that role. +**Row-level security compares instants, not spellings.** A stream row filter on a `DateTime`/`DateTime64` column is compiled and evaluated by the same engine that validates ingest ([internal/typelayer](/access-control#where-each-rule-is-enforced)), so a filter constant in any spelling ClickHouse would accept in a `WHERE` clause matches the stored instant however the payload spelled it, and a constant it cannot read as a timestamp withholds every row for that role (reason `error`). #### Positional formats (CSV / TSV) @@ -321,7 +322,7 @@ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling Cl | `text/csv; header=absent` | `CSV`, strictly positional: header detection is off (`input_format_csv_detect_header=0`), so every line is a record | | `text/csv` (no `header` parameter) | ClickHouse's default `CSV`: header auto-detection stays on | -`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. With no parameter, the positional fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column, and, for a role with column restrictions, minus every column it may not write (an `_eq`-checked column stays) — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column, and only one that meets the [conditions above](#post-v1ingesttabletable--ingest-data). +`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. In either positional reading (no `header` parameter, or `header=absent`), the fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column, and, for a role with column restrictions, minus every column it may not write (an `_eq`-checked column stays) — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column, and only one that meets the [conditions above](#post-v1ingesttabletable--ingest-data). | Body | Outcome | | --- | --- | @@ -331,7 +332,7 @@ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling Cl | too few fields | rejected, code **27** — ClickHouse's own message, e.g. `Cannot parse input: expected ',' before: …` | | too many fields | rejected, code **117** — `Expected end of line` | | a header line, no `header` parameter | ClickHouse detects it and **consumes** it as a header: `total` and every `index` count data rows only | -| a header line, `header=absent` | **not a header** — read as a data row, so it fails to parse (code 27) wherever a column cannot read its own name; the data rows after it still parse | +| a header line, `header=absent` | **not a header** — read as a data row, so it fails to parse wherever a column cannot read its own name, with that column type's own code (72 for a `Float64`, for example); the data rows after it still parse | The messages are ClickHouse's own and differ between ClickHouse lines; branch on the `exception_code`. An empty **TSV** field is the empty string, not a default: `\N` is TSV's `NULL`, which a non-`Nullable` column turns into its default, and a `DateTime64` cannot read `""`. @@ -366,7 +367,7 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ #### Batch Ingest -A JSON array, an NDJSON body, a CSV body or a TSV body ingests a batch in one request. Each record is validated, authorized, deduplicated and published independently, so **one malformed or rejected record never blocks the rest of the batch** — including inside a single-line (compact) JSON array, which WaveHouse re-frames in place before handing it over. An explicit empty array (`[]`) is a valid record-less batch (`200`, `total: 0`) for a role whose insert grant resolves; blank lines in an NDJSON body are skipped. (The SDK's `insert([...])` array helper uses the NDJSON form automatically; every form returns the same response.) +A JSON array, an NDJSON body, a CSV body or a TSV body ingests a batch in one request. Each record is validated, authorized, deduplicated and published independently, so **one malformed or rejected record never blocks the rest of the batch** — including inside a single-line (compact) JSON array declared `application/json`, which WaveHouse re-frames in place before handing it over. The exception is a body ClickHouse's reader cannot get through at all — a record cut off right after a key's `:` or an opening quote, or a single-line array declared NDJSON that holds a bad element: chtypes then declines the whole body, and every record it read answers `422`, with nothing published. ClickHouse's JSON reader also treats a newline as whitespace, so an NDJSON line cut off mid-value can absorb the line after it: the pair is answered as one rejected record, and `total` counts it once. An explicit empty array (`[]`) is a valid record-less batch (`200`, `total: 0`) for a role whose insert grant resolves; blank lines in an NDJSON body are skipped. (The SDK's `insert([...])` array helper uses the NDJSON form automatically; every form returns the same response.) The response counts records read, published, rejected and deduplicated, then lists per-record outcomes: each entry mirrors the single-object response (`ok` / `duplicate` / `error`) plus its 1-based `index`, and carries ClickHouse's numeric `exception_code` when the rejection was its parser's. `results` is truncated to the first 10,000 entries; the four counts stay authoritative. @@ -402,13 +403,15 @@ A `200` is returned whenever the body was read and the records were processed | Status | Body | Cause | | ------ | ---- | ----- | -| 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes. A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | -| 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26 or 27) | +| 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes, or only whitespace (only the first 512 bytes are looked at, so a body opening with 512 bytes of whitespace counts as empty too). A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | +| 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26, 27 or 33) — or, cut right after a key's `:` or an opening quote, a `422` decline of every record in the body | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | | 400 | `{"error":"invalid json: content after the closing ']' of the json array"}` | A body declared `application/json` opening with `[` has something other than whitespace after its closing `]`. That tail is not a record of the array, so the whole request fails and nothing is published | +| 400 | `{"error":"Unknown field found in format header: 'x' at position 1 …","code":"clickhouse.rejected","exception_code":117}` | A `header=present` body whose header names a column the table — or the role's writable set — does not have, or names one twice; see the [single-record table](#post-v1ingesttabletable--ingest-data). Nothing is published | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (same auth gate as the single-object path; surfaces the token reason) | | 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | The resolved role lacks `insert` on the table (checked once, before any record) | | 403 | `{"error":"insert permissions were not resolved for this request"}` | The grant that resolved for the request is not an insert grant; see the [single-record table](#post-v1ingesttabletable--ingest-data). Fails the whole request, an empty array included | +| 404 | `{"error":"unknown table: ..."}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | | 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither a record's fault nor an unavailable tenant; logged. Nothing was published | @@ -418,6 +421,7 @@ A `200` is returned whenever the body was read and the records were processed | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`), mid-batch. Only under [`mq.backend: nats`](/deployment#external-nats); the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the failing record's id is left to lapse rather than given back — so `Retry-After` is that record's dedupe lease, rounded up to whole seconds, when it was deduped; a record published un-deduped has no lapsing claim to wait out, so `Retry-After: 5` | | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now; `Retry-After: 5`. Nothing in the window being reserved was published; the windows before it were, and keep their ids | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | A record's dedupe id is held by another request still being published; includes `Retry-After` (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). Nothing in that record's window was published; the windows before it were | +| 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet; `Retry-After: 5`. Decided before the body is read | | 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet, or its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the zone this process opened that line with — the cause is in the server log, not the body. Decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | @@ -429,7 +433,7 @@ A batch aborted partway — a `503` or `500` after some leading records were alr ### `POST /v1/ops/query` — Query ClickHouse -Executes a SQL statement directly against ClickHouse. **WaveHouse proxies the SQL string verbatim to ClickHouse's HTTP interface** — any statement ClickHouse accepts works, including arbitrary DDL/DML/SYSTEM verbs and inline FORMAT directives. Multi-statement input (`SELECT 1; TRUNCATE t`) also works on recent ClickHouse versions where multi-query is enabled by default; older or restrictively-configured servers may reject the second statement with a clear error. Read queries return a JSON array of result rows; mutations/DDL return HTTP 200 with `[]` on success. DateTime columns are ISO-8601 formatted via the upstream `date_time_output_format=iso` setting — the same server-side rendering `/v1/query`, pipes and the stream use, so a `DateTime64(3)` whole-second value returns `.000Z` on all of them; other types are returned as ClickHouse renders them under `FORMAT JSON`. +Executes a SQL statement directly against ClickHouse. **WaveHouse proxies the SQL string verbatim to ClickHouse's HTTP interface** — any statement ClickHouse accepts works, including arbitrary DDL/DML/SYSTEM verbs and inline FORMAT directives. One statement per request: ClickHouse's HTTP interface refuses multi-statement input (`SELECT 1; TRUNCATE t` is `Code: 62 … Multi-statements are not allowed`, answered `400 clickhouse.rejected`), so send each statement as its own request. Read queries return a JSON array of result rows; mutations/DDL return HTTP 200 with `[]` on success. The proxy does not ask ClickHouse to hold its answer until the statement ends, so a statement that fails after ClickHouse began streaming a large result comes back `200` with ClickHouse's partial output followed by its exception text, which is not valid JSON: treat a body that does not parse as a failure. The request body is capped at 16 MiB. DateTime columns are ISO-8601 formatted via the upstream `date_time_output_format=iso` setting — the same server-side rendering `/v1/query`, pipes and the stream use, so a `DateTime64(3)` whole-second value returns `.000Z` on all of them; other types are returned as ClickHouse renders them under `FORMAT JSON`. :::note[Inline `FORMAT` overrides the JSON envelope] ClickHouse's inline `FORMAT` clause (e.g. `SELECT 1 FORMAT CSV` or `… FORMAT Pretty`) takes precedence over the URL-level `default_format=JSON` setting. When the SQL contains an explicit `FORMAT`, the proxy forwards ClickHouse's raw response body (CSV, Pretty, TSV, …) and passes through the upstream `Content-Type` header — `text/csv`, `text/tab-separated-values`, etc. — so consumers see the right MIME type. The "extract the `data` array" behavior only applies when ClickHouse returned the `FORMAT JSON` envelope, which is the default. @@ -485,11 +489,12 @@ The earlier handler accepted a `params` array bound to `?` placeholders; the HTT | 400 | `{"error":"invalid ?tenant: …"}` / `{"error":"invalid query string: …"}` | The query string does not parse (`?tenant=acme;x=1`, a bad `%` escape), or `tenant` is empty, repeated, or not a tenant id — parsed as strictly as on the [pipe reads](#get-v1opspipes--list-named-pipes) | | 404 | `{"error":"unknown tenant: "}` | No such tenant | | 503 | `{"error":"tenant settings are invalid"}` | The tenant's settings folder was rejected | -| 400 | `{"error":"invalid json"}` | Malformed request body | +| 400 | `{"error":"invalid json"}` | Malformed body, a field other than `sql` (such as the retired `params` array), or more than one JSON value | | 400 | `{"error":"missing sql"}` | Missing `sql` field | -| 400 / 403 / 502 / 503 | `{"error":"","code":"clickhouse.…","retryable":…}` | ClickHouse failed the statement. The status and `code` come from the exception code, not ClickHouse's HTTP status — see [ClickHouse errors on the query paths](#clickhouse-errors-on-the-query-paths). The `error` is ClickHouse's own text verbatim, e.g. `Code: 60. DB::Exception: Table default.x does not exist. (UNKNOWN_TABLE)` | +| 400 / 403 / 502 / 503 | `{"error":"","code":"clickhouse.…","retryable":…}` | ClickHouse failed the statement. The status and `code` come from the exception code, not ClickHouse's HTTP status — see [ClickHouse errors on the query paths](#clickhouse-errors-on-the-query-paths). The `error` is ClickHouse's own text verbatim, e.g. `Code: 60. DB::Exception: Unknown table expression identifier 'x' in scope SELECT * FROM x. (UNKNOWN_TABLE)` | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | The request carried a present-but-invalid/expired token and was denied for lacking permission (the gate surfaces the token reason) | | 403 | `{"error":"forbidden"}` | Caller's role is not the policy `admin_role` (`"admin"` by default) | +| 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap, the data-plane cap ingest uses | | 502 | `{"error":"","code":"clickhouse.unknown","retryable":true}` | An answer with no ClickHouse exception code that is not an outage — a `500` from something in front of ClickHouse. A codeless redirect or `4xx` such as a `404` (a wrong path) is `502 clickhouse.misconfigured`, `retryable: false`; a codeless `408`/`429` is `503 clickhouse.unavailable`, a `413` is `400 clickhouse.rejected` | | 503 | `{"error":"clickhouse request failed: ...","code":"clickhouse.unavailable","retryable":true}` | ClickHouse could not be reached, or the query timed out (connection refused, the upstream went away mid-request); `Retry-After: 5`. A TLS failure, such as an untrusted certificate, is `502 clickhouse.unknown` | | 502 | `{"error":"clickhouse response exceeded N bytes; ...","code":"clickhouse.response_too_large","retryable":false}` | Response body exceeded the 64 MiB memory-safety cap. Narrow the query, add a `LIMIT`, or use `FORMAT JSONEachRow` with a streaming client outside WaveHouse. | @@ -547,14 +552,14 @@ Every column the query references — in `columns`, an aggregation argument, `fi | `group_by` | string[] | No | GROUP BY columns. | | `order_by` | object[] | No | ORDER BY clauses (`column`, `dir`). | | `limit` | int | No | Max rows. Omitted or above the configured `query.default_max_rows` (default 10,000) → silently capped at that value; a policy `max_rows` can lower it further (see [Access Control](/access-control#resource-limits)). | -| `time_range` | object | No | Time window (`column`, `since`, `until`). `since`/`until` accept RFC3339 or Go-duration relative values ("1h", "30m", "7d", "2w" — day and week suffixes expand to hours). Relative values mean that long *ago*. Both bounds are instants: they reach ClickHouse as RFC 3339 in UTC, so the column's zone cannot shift them. The window applies only when `column` and `since` are set — an `until` without `since` is ignored. | +| `time_range` | object | No | Time window (`column`, `since`, `until`). `since`/`until` accept RFC3339 or Go-duration relative values ("1h", "30m", "7d", "2w" — day and week suffixes expand to hours). Relative values mean that long *ago*. Both bounds, absolute ones included, are truncated down to the tenant's `query.timestamp_bucket_seconds` (default 60; `0` disables) so near-identical queries share a cache entry, then reach ClickHouse as RFC 3339 in UTC, so the column's zone cannot shift them. The window applies only when `column` and `since` are set — an `until` without `since` is ignored. | :::note[Filter values bind as strings] -Every bound value — a caller's filter, a policy row filter, an insert `check` — binds as a `{p:String}` parameter and is compared under the column's own type. One rule across all three surfaces, and the answer is the server's: send ClickHouse's own spelling for a value and it reads it. A policy claim on an integer column is compared through a strict cast on top, so a claim that is not the canonical spelling of a value the column can hold matches nothing instead of wrapping (see [Access Control](/access-control#jwt-claim-templating)); a caller's own filter keeps the plain form, since it can only narrow what the policy admits. +Every bound value — a caller's filter, a policy row filter, an insert `check` — binds as a `{p:String}` parameter and is compared under the column's own type. One rule across all three surfaces, and the answer is the server's: send ClickHouse's own spelling for a value and it reads it. A policy claim on an integer column is compared through a strict cast on top, so a claim that is not the canonical spelling of a value the column can hold matches nothing instead of wrapping (see [Access Control](/access-control#jwt-claim-templating)); a caller's own filter keeps the plain form, since it can only narrow what the policy admits. On a non-integer column there is no strict cast, so a policy claim the type cannot read (`abc` on a `Decimal` or a `Float`) fails the whole read with `400 clickhouse.rejected` (code 53 or 72), where the stream withholds the row with reason `error`. -On **this endpoint**, a caller's filter on a `Date`, `DateTime` or `DateTime64` column (`Nullable` or `LowCardinality` too) is parsed by ClickHouse rather than compared as text, because compared directly ClickHouse refuses RFC 3339 on those columns. The value still reaches the server as you wrote it, inside `parseDateTime64BestEffort(…)` with the column's declared zone: an offset or `Z` is the exact instant, a zone-less `"2026-06-21 13:00:00"` is local time in the column's zone (else the server's), a Unix-seconds number works, and a fraction compares exactly, so `> "…04:00:00.5Z"` on a whole-second column excludes `04:00:00`. A `Date` column takes the date of that instant in the server's zone. A value ClickHouse cannot parse is `400 clickhouse.rejected`. `like` compares text and is never parsed. A policy row filter keeps the plain form, so the query path and the stream judge it alike. +On **this endpoint**, a caller's filter on a `Date`, `DateTime` or `DateTime64` column (`Nullable` or `LowCardinality` too) is wrapped in a ClickHouse parse rather than left to the comparison's own cast: compared directly, a `Date` column refuses an RFC 3339 value (code 53), a `DateTime` column truncates a fraction to whole seconds, and whether RFC 3339 is read at all depends on the server's `cast_string_to_date_time_mode` (26.8's default `best_effort` reads it, `basic` refuses it). The value still reaches the server as you wrote it, inside `parseDateTime64BestEffort(…)` with the column's declared zone: an offset or `Z` is the exact instant, a zone-less `"2026-06-21 13:00:00"` is local time in the column's zone (else the server's), a Unix-seconds number works, and a fraction compares exactly, so `>= "…04:00:00.5Z"` on a whole-second column excludes `04:00:00`. A `Date` column takes the date of that instant in the server's zone. A value ClickHouse cannot parse is `400 clickhouse.rejected`. `like` compares text and is never parsed. A policy row filter keeps the plain form, so the query path and the stream judge it alike. -An `in` list travels as a ClickHouse **external table**, not a query parameter, so no ClickHouse field limit applies to it: any list the 1 MiB request body can carry reaches the server. Each element is converted to the column's type (`accurateCastOrNull`, or the timestamp parse above), so `["12.5"]` matches a `Decimal` 12.50, and an element that is not a value of the column's type (`"256"` on a `UInt8`, `"1.5"` on an integer column) matches no row. Scalar values ride on the request line, where ClickHouse takes at most 128 KiB per value once URL-encoded and about 1 MiB for the whole line. With an `in` list the SQL itself travels in one 128 KiB form field. Past any of these the request is a `400` that names the limit, before anything is sent. +An `in` list travels as a ClickHouse **external table**, not a query parameter, so no ClickHouse field limit applies to it: any list the 1 MiB request body can carry reaches the server. Each element is converted to the column's type (`accurateCastOrNull`, or the timestamp parse above), so `["12.5"]` matches a `Decimal` 12.50, and an element that is not a value of the column's type (`"256"` on a `UInt8`, `"1.5"` on an integer column) matches no row — except on a `Date` or `DateTime` column, where an element the parse cannot read fails the whole query with `400 clickhouse.rejected`. Scalar values ride on the request line, where ClickHouse takes at most 128 KiB per value once URL-encoded and about 1 MiB for the whole line. With an `in` list the SQL itself travels in one 128 KiB form field. Past any of these the request is a `400` that names the limit, before anything is sent. ::: :::note[Identifier names] @@ -584,12 +589,17 @@ The inbound request body is capped at 1 MiB; a body over the cap is rejected wit | Status | Body | Cause | | ------ | ---- | ----- | +| 400 | `{"error":"missing table"}` | No `table` query parameter | +| 400 | `{"error":"invalid json"}` | The body is not a structured query (malformed JSON, or a field of the wrong type such as a non-string entry in `columns`) | +| 400 | `{"error":"columns and select_all are mutually exclusive"}` / `{"error":"select_all cannot be combined with aggregations"}` | Both projection forms, or `select_all` with aggregations | | 400 | `{"error":"unknown column: x"}` | Schema validation error — an unknown column, a bad aggregation, or an unparseable `time_range` `since`/`until` (neither a relative duration nor an RFC3339 timestamp) | | 400 | `{"error":"filter value must not be null"}` | A filter carries `"value": null`. It is refused rather than answered: `col = NULL` is never true, and an empty parameter would silently ask a different question | -| 400 | `{"error":"filter value too large: …"}` / `{"error":"query too large: …"}` | A scalar filter value over the 128 KiB ClickHouse's HTTP interface takes for one value once URL-encoded, all of them over the request line, or a query with an `in` list whose SQL is over the 128 KiB form field it travels in. An `in` list itself is never refused for size | +| 400 | `{"error":"filter value too large: …"}` / `{"error":"filter values too large: …"}` / `{"error":"query too large: …"}` | A scalar filter value over the 128 KiB ClickHouse's HTTP interface takes for one value once URL-encoded, all of them over the request line, or a query with an `in` list whose SQL is over the 128 KiB form field it travels in. An `in` list itself is never refused for size | +| 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid or expired token was supplied and the request was then denied (the gate surfaces the token reason) | | 403 | `{"error":"forbidden"}` | Role lacks select permission on table | | 403 | `{"error":"column \"x\" not allowed"}` | Column denied by policy | | 403 | `{"error":"aggregation \"x\" not allowed"}` | Aggregation fn denied by policy | +| 403 | `{"error":"no columns readable for role"}` | `select_all` from a role that may read no column of the table | | 404 | `{"error":"unknown table: x"}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 1048576 bytes"}` | Request body over the 1 MiB cap | | 400 / 403 / 502 / 503 | `{"error":"Code: 60. DB::Exception: …","code":"clickhouse.…","retryable":…}` | ClickHouse failed the query: a column dropped since the schema was discovered (`400 clickhouse.rejected`), the role's `max_rows_to_read`/`max_memory_usage` cap, or a `max_execution_time` no longer than `clickhouse.query_timeout` (`400 clickhouse.limit_exceeded`), a response over the 64 MiB read cap (`502 clickhouse.response_too_large`; earlier versions had no cap here), ClickHouse down (`503 clickhouse.unavailable`, `Retry-After: 5`), … — see [ClickHouse errors on the query paths](#clickhouse-errors-on-the-query-paths) | @@ -646,7 +656,7 @@ Opens a persistent SSE connection for real-time event streaming. Supports histor | Param | Type | Default | Description | | ----- | ---- | ------- | ----------- | | `table` | string | (required) | Table name to subscribe to. Returns `400` only if missing/empty; other values aren't rejected — the name is encoded into a NATS-safe subject token (wildcards `*` / `>` are percent-encoded), so a nonexistent or odd name simply matches no events. | -| `since` | string | — | RFC 3339 or RFC 3339 Nano timestamp. If provided, replays historical events from NATS before switching to live streaming. | +| `since` | string | — | RFC 3339 or RFC 3339 Nano timestamp. If provided, replays historical events from NATS before switching to live streaming. A value that does not parse is ignored: no replay, and no `400`. | | `token` | string | — | JWT token (alternative to `Authorization` header, useful for `EventSource`). Stripped from URL after extraction. | **Headers:** @@ -657,7 +667,7 @@ Opens a persistent SSE connection for real-time event streaming. Supports histor **Response:** SSE stream (`text/event-stream`). Data events include an `id:` field set to the event's `received_timestamp`. The stream opens with a `: connected` comment and emits a minimal `:` keepalive comment periodically (every 30 seconds by default), which keeps a quiet connection from being closed by a proxy; both are standard SSE comments that `EventSource` ignores (raw consumers should skip `:`-prefixed lines). When the server stops (see [Stopping](/deployment#stopping)) it ends every open stream immediately rather than holding it for the drain; `EventSource` reconnects on its own and resumes from `Last-Event-ID`. A reload that stops serving the stream's tenant — its folder removed or rejected, over a [nested settings directory](/deployment#the-nested-settings-directory) — ends that tenant's open streams the same way, and the reconnect then gets its `404` (removed) or `503` (rejected): the SDK stops on the `404` and retries the `503`, resuming from `Last-Event-ID` once the folder is back, while a browser `EventSource` treats either as fatal. A browser going cross-origin reads either refusal only when it passes CORS: it is decorated from tenant `0`'s list ([multi-tenant deployments](/deployment#multi-tenant-deployments)), so where tenant `0` is not served or its list does not admit the page's origin, the SDK sees a network error instead and keeps re-dialing. -**Row values arrive positionally, and the column names are announced separately.** Before the first row, and again whenever the column list changes, the stream sends an `event: schema` frame naming the columns of the rows that follow — in order, already reduced to what the caller's role may read. That re-announcement is **not** guaranteed after a gap-fill across a column change; see the arity note below. Every data frame's `row` array then has exactly one value per announced column, in that order. `schema` is a **named** SSE event, so a browser `EventSource` must `addEventListener('schema', …)` — it never reaches `onmessage`. A schema frame carries **no** `id:` line, so it never moves the client's `Last-Event-ID`. In the example below the table has its own `received_timestamp` **column**, which collides by name with the frame's top-level `received_timestamp` **field** — they are different values: the field is when WaveHouse received the event (WaveHouse's own RFC 3339 timestamp), the row slot is that column as ClickHouse rendered it (a record that omitted it carries the evaluated `DEFAULT`, not `null` — see [Timestamp rendering](#timestamp-rendering)). +**Row values arrive positionally, and the column names are announced separately.** Before the first row, and again whenever the column list changes, the stream sends an `event: schema` frame naming the columns of the rows that follow — in order, already reduced to what the caller's role may read. That re-announcement is **not** guaranteed after a gap-fill across a column change; see the arity note below. Every data frame's `row` array then has exactly one value per announced column, in that order. `schema` is a **named** SSE event, so a browser `EventSource` must `addEventListener('schema', …)` — it never reaches `onmessage`. A schema frame carries **no** `id:` line, so it never moves the client's `Last-Event-ID`. In the example below the table has its own `received_timestamp` **column**, which collides by name with the frame's top-level `received_timestamp` **field** — they are different values: the field is when WaveHouse received the event (WaveHouse's own RFC 3339 timestamp), the row slot is that column as ClickHouse rendered it (a record that omitted it carries the evaluated `DEFAULT`, not `null` — see [the `row` field](#internal-wire-format-nats)). ```text event: schema @@ -670,13 +680,13 @@ id: 2026-03-24T12:00:01.456Z data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,"2026-03-24T12:00:01.456Z"]} ``` -A raw consumer must keep the most recent announced column list and zip each `row` against it; a column the record omitted still has its slot, holding its evaluated `DEFAULT` (or the type's default — `null` only on a `Nullable` column with none), so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you and still yields row objects — `.stream()` and `.liveQuery()` are unchanged. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. +A raw consumer must keep the most recent announced column list and zip each `row` against it; a column the record omitted still has its slot, holding its evaluated `DEFAULT` (or the type's default — `null` only on a `Nullable` column with none), so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you: `.stream()` and `.liveQuery()` zip each row into an object. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. Each SSE connection is bound to a single `?table=`; to consume multiple tables, open one connection per table. -Values of top-level `DateTime`/`DateTime64` columns inside `row` are ClickHouse's own rendering of the stored value — the exact bytes chtypes' `RowsExport` produced for that record (see [Timestamp rendering](#timestamp-rendering)), not a WaveHouse rewrite — so a live event and a `/v1/query` read of the same row agree on spelling **by construction**, with no separate canonicalization step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). A column declared with a non-UTC zone is still rendered in UTC (the `Z` form), so the strings compare as the instants do. +Values of `DateTime`/`DateTime64` columns inside `row`, nested ones included, are ClickHouse's own rendering of the stored value — the exact bytes chtypes' `RowsExport` produced for that record (see [Timestamp rendering](#timestamp-rendering)), not a WaveHouse rewrite — so a live event and a `/v1/query` read of the same row agree on spelling **by construction**, with no separate canonicalization step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). A column declared with a non-UTC zone is still rendered in UTC (the `Z` form), so the strings compare as the instants do. -**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause — a connection is never delivered a row the query path would hide for that role, and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable`, while other tenants' streams (and roles with no row filter) are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant's line is not served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert ClickHouse later rejects and parks on the dead-letter queue — a data-shape problem chtypes did not catch, since an outage only delays a row and never drops it — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), but an expired token or changed claims take effect only when the client reconnects. +**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause — a connection is never delivered a row the query path would hide for that role, and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable`, while other tenants' streams (and roles with no row filter) are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant's line is not served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` or `ALIAS` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert ClickHouse later rejects and parks on the dead-letter queue — a record the in-process parse accepted that the insert did not (a schema change between publish and insert, for instance), since an outage only delays a row and never drops it — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), but an expired token or changed claims take effect only when the client reconnects. **CORS:** `/v1/stream` honors the request's tenant's `cors.allowed_origins` allowlist (settings directory) like every endpoint — the preflight included, which a browser sends without `X-Tenant-ID`, so over [a nested settings directory](/deployment#multi-tenant-deployments) the fronting proxy has to set the header on the `OPTIONS` too. Note that a **header-authenticated stream preflights before it connects** — `Authorization` is not CORS-safelisted — where a bare `EventSource` never preflighted at all: its request is not a `fetch()`, so Fetch's unsafe-request flag is never set and `Last-Event-ID` rides on the plain `GET`. Both headers are allow-listed, so an allowed origin connects *and* resumes cross-origin. @@ -807,7 +817,7 @@ Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant] | Param | Type | Default | Description | | ----- | ---- | ------- | ----------- | | `tenant` | string | `0` | The tenant whose dead-letter queue is read. | -| `table` | string | — | Filter stats to a specific table name (e.g., `?table=clicks` returns only the `clicks` count). | +| `table` | string | — | Filter stats to a specific table name (e.g., `?table=clicks`): `tables` then holds only that entry, while `total` stays the tenant's whole dead-letter count. | **Response:** @@ -850,7 +860,7 @@ Returns a specific named pipe definition: "sql": "SELECT page, count() as views FROM clicks WHERE received_timestamp >= {{start_date}} GROUP BY page LIMIT {{limit}}", "parameters": [ {"name": "start_date", "type": "string", "required": true}, - {"name": "limit", "type": "number", "required": false, "default": 100} + {"name": "limit", "type": "number", "default": 100} ], "description": "Top pages by view count", "allowed_roles": ["viewer"] @@ -922,7 +932,7 @@ Three values, where the envelope above has four: this is the frame a role restri ## Dead Letter Queue (DLQ) -When ClickHouse **rejects** a batch insert (a value it cannot parse, a type mismatch, a table or column it does not have), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A ClickHouse that **cannot take** the insert — down, unreachable, timing out, overloaded, read-only, or refusing WaveHouse's credentials — never sends a row here: the batch stays in the tenant's ingest queue and is retried with backoff until it inserts (see [Ingest Pipeline](/ingest-pipeline#when-clickhouse-cannot-take-an-insert)). A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: malformed JSON, an envelope of an unknown `format`, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values exactly as published: ClickHouse's own rendering of the stored value, since chtypes already validated and coerced the record before it was ever published — see [Timestamp rendering](#timestamp-rendering)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. Because chtypes catches the type and shape problems synchronously at ingest, a row that reaches this DLQ path is one ClickHouse later rejected for a reason chtypes couldn't have caught up front, not a data mismatch; an outage only delays a row and never parks it. +When ClickHouse **rejects** a batch insert (a value it cannot parse, a type mismatch, a table or column it does not have), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A ClickHouse that **cannot take** the insert — down, unreachable, timing out, overloaded, read-only, or refusing WaveHouse's credentials — never sends a row here: the batch stays in the tenant's ingest queue and is retried with backoff until it inserts (see [Ingest Pipeline](/ingest-pipeline#when-clickhouse-cannot-take-an-insert)). A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: malformed JSON, an envelope of an unknown `format`, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values exactly as published: ClickHouse's own rendering of the stored value, since chtypes already validated and coerced the record before it was ever published — see [Timestamp rendering](#timestamp-rendering)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. Because chtypes catches the type and shape problems synchronously at ingest, a row that reaches this DLQ path is one ClickHouse later rejected rather than a record the in-process parse would have refused (a schema change between publish and insert, for instance); an outage only delays a row and never parks it. Under [`mq.backend: nats`](/deployment#external-nats) the parked rows of every tenant go to one shared dead-letter stream instead, under `.dlq.{tenant}.{table}`; the bodies and headers are the same. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index a8c1e423..10b6a992 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -23,7 +23,8 @@ flowchart TD subgraph api["WaveHouse API Layer"] IH["Ingest Handler"] --> SR["Schema Registry"] - SR --> DD["Dedupe (optional)"] + SR --> TL["Type layer
(ClickHouse's parser, chtypes)"] + TL --> DD["Dedupe (optional)"] DD --> MQ["MQ (NATS)"] MQ --> BC["Buffer Consumer
(batch flush)"] BC -.->|rejected rows| DLQ["DLQ"]:::fail @@ -45,7 +46,7 @@ flowchart TD ## Binaries -WaveHouse ships a single binary, `wavehouse`: an all-in-one process running the API, batch worker, embedded NATS JetStream, and optional embedded Pebble dedup. The binary also loads a second artifact at start: a per-ClickHouse-version shared library (`internal/typelayer`, via [chtypes](/deployment#chtypes-artifacts)) that runs ClickHouse's own parser in-process for ingest validation, type coercion, and row-level security, and only a process running the `api` role loads it. The binary requires cgo (dlopen only — no static link to the artifact) and, on Linux, glibc 2.34 or later, so supported platforms are Linux amd64/arm64 and macOS arm64. The only external network dependency is ClickHouse, unless a shared backend is selected: `cache.backend: redis`, `dedupe.backend: dynamodb`, or `mq.backend: nats`, which points the queue at a NATS cluster the operator runs and lets several processes, each running some of the [roles](/configuration#process-roles), share it. `cmd/wavehouse` is the shell — subcommand dispatch, the logger, `config.Load`, the signal context — and `internal/app` is the process itself (see [`app/`](#app--process-wiring) below). +WaveHouse ships a single binary, `wavehouse`: an all-in-one process running the API, batch worker, embedded NATS JetStream, and optional embedded Pebble dedup. An `api`-role process also uses a second artifact: a per-ClickHouse-line shared library (`internal/typelayer`, via [chtypes](/deployment#chtypes-artifacts)) that runs ClickHouse's own parser in-process for ingest validation, type coercion, and row-level security, opened the first time a tenant on that line is bound; an `api` process with no artifact installed refuses to boot, and no other role needs one. The binary requires cgo (dlopen only — no static link to the artifact) and, on Linux, glibc 2.34 or later, so supported platforms are Linux amd64/arm64 and macOS arm64. The only external network dependency is ClickHouse, unless a shared backend is selected: `cache.backend: redis`, `dedupe.backend: dynamodb`, or `mq.backend: nats`, which points the queue at a NATS cluster the operator runs and lets several processes, each running some of the [roles](/configuration#process-roles), share it. `cmd/wavehouse` is the shell — subcommand dispatch, the logger, `config.Load`, the signal context — and `internal/app` is the process itself (see [`app/`](#app--process-wiring) below). ## Internal Packages @@ -84,9 +85,9 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`sql_classify.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. -- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (anything but whitespace after the closing `]` is a whole-request `400 invalid json: content after the closing ']' of the json array`, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read and ahead of every other tenant's traffic. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's two-minute duplicate window — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). +- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (brackets that do not balance are a whole-request `400 invalid json: unterminated json array`, anything but whitespace after the closing `]` a whole-request `400 invalid json: content after the closing ']' of the json array`, and a `…WithNames` header naming an unknown or repeated column a whole-request `400 clickhouse.rejected` with code 117, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read, without affecting other tenants. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's duplicate window (two minutes on the embedded broker) — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso`, the same spelling the structured-query path and the SSE wire use, so a timestamp reads the same on every surface. -- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `date_time_output_format=iso`, so a timestamp is RFC 3339 in UTC and matches the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so the same classification and `code` table as the native path applies, and a response past the 64 MiB cap is `clickhouse.response_too_large`. The read paths hold at most the tenant's pool size in HTTP connections — its `max_open_conns`, or the largest among the tenants sharing its connection tuple (address, database, user, password, TLS) — one cap per tuple, like the native pool; tenants on one server with a different database, user, password or TLS settings each have their own. +- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `output_format_json_named_tuples_as_objects=1`, `date_time_output_format=iso`, so a timestamp is RFC 3339 in UTC and matches the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so `chconn.Classify` and `writeCHError` apply as on every other ClickHouse path, and a response past the 64 MiB cap is `clickhouse.response_too_large`. The read paths hold at most the tenant's pool size in HTTP connections — its `max_open_conns`, or the largest among the tenants sharing its connection tuple (address, database, user, password, TLS) — one cap per tuple, like the native pool; tenants on one server with a different database, user, password or TLS settings each have their own. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) chconn.Target` beside it — each resolves the request's tenant per call, and the zero target (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for one tenant's per-table parked counts (optionally one table) and its total — the tenant `?tenant=` names, read strictly by `opsTenant`, tenant `0` without it. The tenant is looked up in the MQ, not the settings registry, so a rejected or removed tenant's parked rows are read like a served one's; a tenant with no dead-letter queue (`mq.ErrNoDeadLetterQueue`) is a 404, and any other failure to read it a 500. The queue itself is `internal/mq`'s. @@ -94,7 +95,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request ### `app/` — Process wiring -- **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability (after which each `Config.Warnings` line is logged at `WARN`), ClickHouse pools, schema discovery, the dedupe stores, the MQ (embedded NATS with its ingest + DLQ streams, or the external NATS), cache, the lease coordinator, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. The boot config's `roles` decide which of them a process wires: every process gets the settings registry, observability, the MQ, the coordinator, the reload triggers and a listener; `api` adds schema discovery, the dedupe stores, streaming, auth and the full router; `ingest` adds the ingest worker; `sweeper` adds the sweeper on `mq.backend: embedded` only (under `nats` the streams' retention replaces it); the ClickHouse pools and the cache come with `api` or `ingest`. A process without `api` serves `api.NewOpsRouter` (probes, `/version`, the metrics path, and the settings reload behind the operator key alone, `wireOpsAuth`) on `server.port`. `config.Validate` refuses a role set the backends cannot serve (a split over the embedded MQ, ingest on a shared MQ over a local coordinator, a sweeper-only process on a shared MQ, or `api` without `ingest` and the reverse over a local cache), and `New` refuses a `Config` with no roles, which only one built without `config.Load` can have. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. +- **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability (after which each `Config.Warnings` line is logged at `WARN`), ClickHouse pools, the type layer (the chtypes registry), schema discovery, the dedupe stores, the MQ (embedded NATS with its ingest + DLQ streams, or the external NATS), cache, the lease coordinator, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. The boot config's `roles` decide which of them a process wires: every process gets the settings registry, observability, the MQ, the coordinator, the reload triggers and a listener; `api` adds the type layer, schema discovery, the dedupe stores, streaming, auth and the full router; `ingest` adds the ingest worker; `sweeper` adds the sweeper on `mq.backend: embedded` only (under `nats` the streams' retention replaces it); the ClickHouse pools and the cache come with `api` or `ingest`. A process without `api` serves `api.NewOpsRouter` (probes, `/version`, the metrics path, and the settings reload behind the operator key alone, `wireOpsAuth`) on `server.port`. `config.Validate` refuses a role set the backends cannot serve (a split over the embedded MQ, ingest on a shared MQ over a local coordinator, a sweeper-only process on a shared MQ, or `api` without `ingest` and the reverse over a local cache), and `New` refuses a `Config` with no roles, which only one built without `config.Load` can have. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. - **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. A layer with a choice of implementation — `wireMQ`, `wireCache`, `wireDedupe`, `wireCoord` — picks it there and nowhere else, in a `switch` on the boot config's `.backend` with one case per backend; the default case refuses boot, which only a `config.Config` built without `config.Load` reaches, since `Validate` refuses a value no case handles. `wireCache` has two: `local`, the in-process `LocalCache`, and `redis`, the shared `RedisCache` built from the `cache.redis` block, which boots bypassed rather than failing when its server is unreachable. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where each would be redelivered every ack wait for as long as the tenant is away and would hold that tenant's ack floor, so the sweeper could purge none of its queue past it. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`registryFor`, `chTargetFor`, `queryTimeout`, and `readConns`, the cap on the HTTP connections a tenant's pipes and structured queries hold), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is the zero `chconn.Target`, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the whole cache — structured-query and pipe results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultPolicy` reads it through the registry, where a flat directory's tenant `0` is always served. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth, dedupe and cache hooks use too), ending the open streams of a tenant no longer served, and `wireCache`'s hook prunes the cache's version index the same way (`LocalCache.Prune`, [#262](https://github.com/Wave-RF/WaveHouse/issues/262)), so a tenant no longer served stops holding it. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The ingest worker's queue, over a `mq.Sharded` broker (`mq.backend: nats`), is wrapped in `ingest.ClaimShards` over the same coordinator, so each process consumes only its share of the shards; the `nats` coordinator holds the ingest processes' membership leases there, since no sweeper is wired under `nats`. The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own, and runs only while its process holds the `sweeper` lease (`elected`, which wraps `coord.RunElected` over the coordinator `wireCoord` opens: `local` keeps leases in the process, so the one process always holds it; `nats` calls `ExternalNATS.Leases` on the MQ's own connection with the bucket `coordBucket` names — `coord.nats.bucket`, or `mq.DefaultNATSCoordBucket` of the subject prefix — and `instance_id` as the holder. `wireNATSMQ` hands the same bucket name to the topology, so boot waits for it with the streams. The coordinator is added after the MQ, so it closes first and resigns its terms while the connection is still up). The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe`'s `pebble` case builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) and then its table — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. `wireDedupe`'s `dynamodb` case is `wireDynamoDedupe`, in wire_dynamodb.go (below). `wireHTTP` hands the ingest handler `dedupe.lease` (`IngestHandler.DedupeLease`) whichever backend is chosen. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ`'s `nats` case builds an `mq.NATSConfig` from the boot config's `mq.nats` block and calls `mq.NewNATS`, which waits for the operator's topology under `New`'s context; it hands over no budget, since the operator's streams set every limit. Both cases end in `adoptMQ`, which registers the MQ's close and the system gauges. The `embedded` case hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. - **wire_dynamodb.go** — `wireDedupe`'s `dynamodb` case, split out of wire.go so the e2e suite's coverage exclude for it (the e2e binary always runs Pebble dedupe, never DynamoDB) doesn't have to blanket wire.go itself: builds the same `dedupe.Stores` over `Dynamo.Tenant`, gated (`Factory.Gated`) on the table's check: boot runs `Dynamo.Check` (after `CreateTable`, when `dedupe.dynamodb.create_table` is on) whether or not any tenant has dedupe on. Boot is refused only for a misconfigured table (an error that is not `ErrUnavailable`) over a flat directory whose tenant has dedupe on; every other failure boots with the switched-on stores closed, the check retried until it passes by a background component that backs off from one second to thirty (a nested directory has no watcher, and a flat one's table can come good with no settings change). The `AfterAdopt` hook never runs the check, since it holds the lock that serializes reloads, and it does not wait on a tenant whose `dedupe.enabled` is unchanged either — `Managed.Apply`'s no-op fast path settles that case under its own read lock, so the hook only takes a store's write lock, and so waits for that tenant's in-flight `Reserve`/`Commit`/`Release` calls to finish, on a genuine flip. It applies every store against the last check's result, so a tenant a reload switches on fails closed meanwhile, and wakes the retry, so a reload still retries at once. It has no Pebble gauges. @@ -105,7 +106,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)) lives next to the keepalive primitives it shares. One abstraction per file. -- **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row. Visibility itself is decided by `internal/typelayer` (`Table.ParseRow` once per event, on the event's tenant's table set, then `Row.Visible` per subscriber) — the same ClickHouse parsing and comparison semantics the server's own `WHERE` clause applies, for every column type, rather than a hand-written per-type comparator; a predicate error, a policy column the table no longer has, a schema drift between the event and the live table, or an engine that is unavailable for that tenant all withhold the row rather than guessing. Each row withheld this way increments `wavehouse_sse_rows_withheld_total{table,role,reason}`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and caching the per-table column-kind lookup across the replay loop. `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. +- **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row. Visibility itself is decided by `internal/typelayer` (`Table.ParseRow` once per event, on the event's tenant's table set, then `Row.Visible` per subscriber) — the same ClickHouse parsing and comparison semantics the server's own `WHERE` clause applies, for every column type, rather than a hand-written per-type comparator; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (one the inserting role cannot write, or a `MATERIALIZED`/`ALIAS` column; reason `decline`, though `/v1/query` still returns the row), a schema drift between the event and the live table, or an engine that is unavailable for that tenant all withhold the row rather than guessing. Each row withheld this way increments `wavehouse_sse_rows_withheld_total{table,role,reason}`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and, under a row filter, preparing one parsed row per replayed event (closed at once, since a gap-fill can run for thousands of events). `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. - **subscriber.go** — `Subscriber`, the per-connection handle. It carries the connection's JWT claims, fixed at construction (`NewSubscriber(claims, metrics)`, no setter) — the claims the `Hub` resolves a role's row-level `filter` against, and immutability is what makes the fan-out's unsynchronized claims read race-free structurally. It owns a single ready-to-write outbound queue of `Frame`s (each tagged with its `kind`, so the handler labels the write where it happens): producers — the keepalive wheel and the event `Hub` — fan frames in with `Send` (non-blocking; a full queue drops, and `Send` itself counts the drop by frame kind, so no producer can forget to), and the handler drains `Frames()` to the client verbatim. The queue is sized for buffering live events (cap 64, up from the keepalive-only cap 1; #152 will make it a knob), and `Evict` closes its `Evicted()` channel, once, for the handler to end the stream: the `Hub`'s `Prune` does for a tenant no longer served, and the slow-consumer follow-up will for a wedged consumer. - **bucket.go** — `Bucket`, the reusable fan-out primitive: a concurrency-safe set of subscribers. `Push` fans one `Frame` to every member fire-and-forget — the keepalive wheel's ring is its only caller now that both `Hub` paths iterate `Snapshot`, since the schema announcement is per connection even where the projection is shared per role; `Snapshot` exposes the members so the event `Hub` can evaluate row visibility per subscriber before sending (drop counting lives in `Send` itself). The `Hub` holds one `Bucket` per `(topic, role)` so a projected frame is built once and sent to every member instead of re-projected per subscriber. - **heartbeat.go** — The keepalive wheel (`Heartbeater`). A single process-wide ticker fans a minimal `:` comment across the ring of `Bucket`s, waking ~1/N of live streams per tick so the writes don't synchronize. The effective per-connection keepalive period is `stream.keepalive_interval` in the settings directory (the wheel ticks every `keepalive_interval ÷ keepalive_buckets`, so one rotation spans the interval; a reload calls `Reconfigure`, which rebuilds the ring in place with every live subscriber carried over); the owning handler goroutine does the actual write, so the shared ticker never touches a `ResponseWriter` directly. @@ -113,12 +114,12 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `auth/` — Authentication -- **auth.go** — `NewAuthenticator(cfg, tenantOf, policies)` owns one verifier per tenant: `cfg` is the boot-config half (the HMAC secret and the operator key, shared by every tenant), and `Reconfigure(id, wiring)` gives a tenant a verifier built from its own `auth` block (`auth.jwks_url` / `auth.role_claim`; see [Settings Directory — Authentication](/settings-directory#authentication)) — key source plus its pinned `alg` allowlist, swapped atomically after a reload that adopts the tenant and kept when the wiring did not change; `Prune` drops the verifiers of tenants that stopped being served (removed or rejected), `Close` stops every JWKS refresh. `Middleware()` reads the request tenant's verifier per request — the tenant of the store `TenantMW` resolved (`settings.Store.Tenant()`, through the injected `TenantSource`), tenant `0` on a tenant-exempt route — so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set; a tenant with no verifier fails closed. A JWKS key set is fetched off the calling goroutine, so neither boot nor a reload waits on the endpoint: the verifier is in place at once and *pending* until a set has been stored (`observedKeys`, the one success signal the library gives — a failed first fetch leaves it holding an empty set). That first fetch is `fetch`'s to retry, from one second doubling to a minute, since a pending verifier never reaches the library and so never triggers its unknown-key-id refetch; once a set is stored the library keeps it fresh on its own, hourly and, rate-limited, on an unknown key id. Each response is capped at 1 MiB. A token checked against a pending verifier records `auth.ErrVerifierPending`, which the router's `refuseUnverifiable` turns into `503` + `Retry-After: 30` on every `/v1` route rather than let the request run under the `default_role`. Verifies JWT tokens with HMAC **or** JWKS (never both), with the accepted `alg` pinned to the active verifier and checked before any key is consulted (rejects `alg: none` and cross-family confusion). Extracts the caller's role from a configurable dot-path claim (`auth.role_claim`, default `role`). Claims parse with `jwt.WithJSONNumber()`, so a numeric claim reaches the policy engine as its exact digits (`json.Number`, never a rounded float64) — part of the row-visibility guarantee (AGENTS.md invariant 12). It always runs and never rejects — a missing/invalid/expired token yields an empty role (resolved to `default_role` downstream), with the token error stashed in context so a denying gate can fail loud (`401`, not a bare `403`). Before the Bearer token it checks a non-JWT operator key (`auth.operator_key`): a constant-time match on the presented credential — an `Authorization: Operator ` header, or the `X-Operator-Key` alias — stamps the live admin role — the request tenant's, read from that tenant's policy (`policies`) — plus an operator bit (`auth.WithOperator`) that `RequireAdmin` honors even under a nil policy — a full-access break-glass credential, audit-logged at Info with no client IP (the audit line goes through the default logger). A presented-but-wrong operator key is logged at `WARN` and counted by `wavehouse_auth_operator_key_failures_total` — a probing signal on the most privileged credential — then falls through like any unauthenticated request (the middleware never rejects). +- **auth.go** — `NewAuthenticator(cfg, tenantOf, policies)` owns one verifier per tenant: `cfg` is the boot-config half (the HMAC secret and the operator key, shared by every tenant), and `Reconfigure(id, wiring)` gives a tenant a verifier built from its own `auth` block (`auth.jwks_url` / `auth.role_claim`; see [Settings Directory — Authentication](/settings-directory#authentication)) — key source plus its pinned `alg` allowlist, swapped atomically after a reload that adopts the tenant and kept when the wiring did not change; `Prune` drops the verifiers of tenants that stopped being served (removed or rejected), `Close` stops every JWKS refresh. `Middleware()` reads the request tenant's verifier per request — the tenant of the store `TenantMW` resolved (`settings.Store.Tenant()`, through the injected `TenantSource`), tenant `0` on a tenant-exempt route — so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set; a tenant with no verifier fails closed. A JWKS key set is fetched off the calling goroutine, so neither boot nor a reload waits on the endpoint: the verifier is in place at once and *pending* until a set has been stored (`observedKeys`, the one success signal the library gives — a failed first fetch leaves it holding an empty set). That first fetch is `fetch`'s to retry, from one second doubling to a minute, since a pending verifier never reaches the library and so never triggers its unknown-key-id refetch; once a set is stored the library keeps it fresh on its own, hourly and, rate-limited, on an unknown key id. Each response is capped at 1 MiB. A token checked against a pending verifier records `auth.ErrVerifierPending`, which the router's `refuseUnverifiable` turns into `503` + `Retry-After: 30` on every `/v1` route rather than let the request run under the `default_role`. Verifies JWT tokens with HMAC **or** JWKS (never both), with the accepted `alg` pinned to the active verifier and checked before any key is consulted (rejects `alg: none` and cross-family confusion). Extracts the caller's role from a configurable dot-path claim (`auth.role_claim`, default `role`). Claims parse with `jwt.WithJSONNumber()`, so a numeric claim reaches the policy engine as its exact digits (`json.Number`, never a rounded float64) — part of the row-visibility guarantee. It always runs and never rejects — a missing/invalid/expired token yields an empty role (resolved to `default_role` downstream), with the token error stashed in context so a denying gate can fail loud (`401`, not a bare `403`). Before the Bearer token it checks a non-JWT operator key (`auth.operator_key`): a constant-time match on the presented credential — an `Authorization: Operator ` header, or the `X-Operator-Key` alias — stamps the live admin role — the request tenant's, read from that tenant's policy (`policies`) — plus an operator bit (`auth.WithOperator`) that `RequireAdmin` honors even under a nil policy — a full-access break-glass credential, audit-logged at Info with no client IP (the audit line goes through the default logger). A presented-but-wrong operator key is logged at `WARN` and counted by `wavehouse_auth_operator_key_failures_total` — a probing signal on the most privileged credential — then falls through like any unauthenticated request (the middleware never rejects). - **context.go** — request-context accessors and their setters for the role, claims, and token error (`RoleFromContext`, `ClaimsFromContext`, `AuthErrorFromContext`, and the matching `With*` helpers). ### `cache/` — Query Cache -- **cache.go** — `Cache` interface: `Lookup`, `Set`, `Invalidate`, `InvalidateTenant`, `Close`, plus `QueryTimeToTTL`, which sets a result's TTL from how long its query took (10 s floor, 1 h ceiling). Every entry is one tenant's: `Lookup` takes the tenant, the caller's query key — `:query:`, built by the two cached handlers in `api/` (`queryCacheKey`, with the tenant read off the request's store — `settings.Store.Tenant`), which use it as their [singleflight](https://pkg.go.dev/golang.org/x/sync/singleflight) key too — and the `Namespace`s the result depends on — one for a structured query, none yet for a pipe (a pipe's table dependencies are [#343](https://github.com/Wave-RF/WaveHouse/pull/343)) — each of that tenant (another tenant's is `ErrForeignDependency`) and naming a table and scope by their raw names, which the cache escapes where it builds a key. `Lookup` returns the `Entry` (a nil value is a miss) and a `Snapshot` of the versions it read; on a miss the handler runs the query and passes that snapshot to `Set`, so a result is filed under the versions read *before* its query ran, and a write that lands while it runs orphans the fill rather than re-homing pre-write rows under the post-write versions ([#382](https://github.com/Wave-RF/WaveHouse/issues/382)). The snapshot is taken before any input a bump invalidates is chosen, the tenant's connection included: a reload that moves the tenant to another address or database runs `Pools.Reconcile` and then `InvalidateTenant` (a repoint that keeps both, such as a username or `tls` change, reads the same tables and bumps nothing), so a request that took the old pool files the old database's rows under a version that bump orphans, whether its `Set` lands before the bump or after. The singleflight leader's snapshot is the one used. `Set` returns an error only when the backend failed; a value the cache declines — larger than it keeps, a non-positive TTL, a zero snapshot — is not one. What a hit, a miss and a bump mean is pinned by the conformance suite every backend runs, `internal/testutil/cachetest`. +- **cache.go** — `Cache` interface: `Lookup`, `Set`, `Invalidate`, `InvalidateTenant`, `Close`, plus `QueryTimeToTTL`, which sets a result's TTL from how long its query took (10 s floor, 1 h ceiling). Every entry is one tenant's: `Lookup` takes the tenant, the caller's query key — `:query:` (the contract, `chRendering`, so a build that renders rows differently never serves another build's entries from a shared cache), built by the two cached handlers in `api/` (`queryCacheKey`, with the tenant read off the request's store — `settings.Store.Tenant`), which use it as their [singleflight](https://pkg.go.dev/golang.org/x/sync/singleflight) key too — and the `Namespace`s the result depends on — one for a structured query, none yet for a pipe (a pipe's table dependencies are [#343](https://github.com/Wave-RF/WaveHouse/pull/343)) — each of that tenant (another tenant's is `ErrForeignDependency`) and naming a table and scope by their raw names, which the cache escapes where it builds a key. `Lookup` returns the `Entry` (a nil value is a miss) and a `Snapshot` of the versions it read; on a miss the handler runs the query and passes that snapshot to `Set`, so a result is filed under the versions read *before* its query ran, and a write that lands while it runs orphans the fill rather than re-homing pre-write rows under the post-write versions ([#382](https://github.com/Wave-RF/WaveHouse/issues/382)). The snapshot is taken before any input a bump invalidates is chosen, the tenant's connection included: a reload that moves the tenant to another address or database runs `Pools.Reconcile` and then `InvalidateTenant` (a repoint that keeps both, such as a username or `tls` change, reads the same tables and bumps nothing), so a request that took the old pool files the old database's rows under a version that bump orphans, whether its `Set` lands before the bump or after. The singleflight leader's snapshot is the one used. `Set` returns an error only when the backend failed; a value the cache declines — larger than it keeps, a non-positive TTL, a zero snapshot — is not one. What a hit, a miss and a bump mean is pinned by the conformance suite every backend runs, `internal/testutil/cachetest`. - **local.go** — `LocalCache`, the in-process L1 on [Ristretto](https://github.com/dgraph-io/ristretto): one pool shared by every tenant (a heavier tenant holds more of it), sized by the boot config's `cache.l1_max_cost`. - **version_manager.go** — `VersionManager`, the invalidation index behind `Invalidate` and `InvalidateTenant`: one version per tenant, per (tenant, table) and per (tenant, table, scope), each keyed by its name alone and bumped in place, so the index holds one entry per live tenant, table and scope however often each is bumped ([#262](https://github.com/Wave-RF/WaveHouse/issues/262)). A query key folds the tenant's version and, for each dependency, its tenant's, table's and scope's, so bumping a table (a scopeless write) orphans every scope of it, and bumping one scope orphans that scope and the whole-table view — scope is reserved and empty today, so every write is the whole-table bump — all without touching the pool. Every field — the caller's query key, the tenant id, and each dependency's table and scope — is escaped and joined by `internal/keyenc` where the key is built, so a dot, a space or a `%` in a name is never read as a separator: each dependency renders as `..
.
..`, and the whole entry key is `|.||…`. The tenant leads every key ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8): the same table under two tenants is two namespaces, so a bump through `Invalidate` under one tenant never touches — and a read under one tenant is never served — the other's results. A tenant's version is a *generation*, unique within the process and handed out by the first key built for the tenant; `BumpTenant` (behind `InvalidateTenant`) drops the tenant's whole index, so the next key gets a fresh generation no cached entry folds, orphaning every cached result of the tenant in one step — a pipe result with no dependencies, and a table no bump ever keyed, included — for a tenant back on a pool after an absence from the fan-out, or moved to another address or database (story 6). `LocalCache.Prune` does the same for every tenant no longer served, which `internal/app` runs after each settings reload, so a tenant removed or rejected stops holding its index. A table bump drops the table's scope versions with it, since every key they were folded into also folds the old table version; and any bump (of a table, a scope or the tenant) under a tenant with no index is a no-op that records nothing, since the next key built for it gets a fresh generation no cached entry folds — so neither the `sharedTables` fan-out nor an insert still in flight for a tenant just pruned brings its index back. The index is per tenant; the cross-tenant invalidation an insert into a shared table needs is not the index's but the wiring's: `internal/app` hands the ingest worker a cache (`sharedTables`) that repeats each bump under every tenant on the same ClickHouse address and database. - **redis.go** — `RedisCache`, the shared backend: one Redis-compatible server (Redis, Valkey, Dragonfly, ElastiCache, MemoryDB — only `GET`, `SET` and `MGET`, no scripts, no client tracking) holds every process's results and versions, so a bump one process makes orphans what every process cached. Versions are random 8-byte tokens in a flat key space, one per tenant (`:{t}:T`), table (`:B:
`) and scope (`:S:
:`, the empty scope being the whole-table view), all under the tenant's hash tag so they share one cluster slot; the table and scope are escaped with `keyenc` after the fixed prefix, so a `:` in a name is never read as the separator. A dependency folds three: the tenant's, its table's and its scope's; a scopeless write bumps `B:
`, a scoped one `S:
:` and `S:
:`, `InvalidateTenant` the tenant's — the lattice `VersionManager` encodes. `Lookup` pipelines an `MGET` of the tokens with a `GET` of the value in one round trip; the value (`:q::`, the hash over their escaped forms, no hash tag, so a tenant's values spread across shards) carries the tokens it was filed under and is a hit only while they are all current. A missing token is created (`SET NX`) and read back, never read as a value, and a token key holding anything but a token (a string of another length, a list, a hash) is replaced, so a token lost to eviction, expiry, `FLUSHALL` or a restart without persistence is a miss for everything under it, never a revival — any `maxmemory-policy` that evicts is safe. Restoring an RDB or AOF snapshot, or a backup, is not a loss but a rollback — a restart after a crash that reloads the server's last save included, which stock Redis and Valkey make by default: the old tokens come back with the values filed under them, so what was invalidated since is served again until its TTL, as after a failover to a replica that missed the bumps. A failure or a timeout past `Timeout` (100 ms) is a miss, a skipped fill and a deferred invalidation; queries never fail on the cache. While this process owes a bump on any of a lookup's tokens, that lookup is a bypass that files nothing: the bump would orphan whatever it found (other processes, which cannot know, serve those entries until it lands). `NewRedis` never fails on an unreachable server: the cache starts bypassed and keeps dialing, each attempt bounded by `DialTimeout` (1 s), and a cluster client's topology read after the handshake by the larger of it and `Timeout`, which is also how long a connection waits on a silent server before it is redialed. Every connection is replaced after a minute (`clientOption`; rueidis retries what was in flight), because one that outlives a failover behind a stable address stays on the demoted node, which answers but refuses writes; the replacement re-resolves the address, so a bypassed process reaches the new primary, and delivers the bumps it owes, within about that long. rueidis dials a replacement lazily, under the context of the operation that lands on it, bounding the dial (TLS included) by `DialTimeout` and then the handshake by `DialTimeout` again, so a reconnect slower than `Timeout` fails that operation. The breaker's probe gives its write twice `DialTimeout` for a reconnect on top of `Timeout`, so such a reconnect still closes an open breaker. The allowance is for a reconnect only: a probe write slower than `Timeout` is repeated under `Timeout`, and the repeat decides, so a server answering slower than `Timeout` stays bypassed. But rueidis spreads commands over several connections (up to four to one server, by `GOMAXPROCS`, and one per cluster node) and the probe reconnects only the one it lands on, so size `Timeout` above a reconnect, or operations that land on the others keep failing. `cache.backend: redis` selects it: `internal/app`'s `wireCache` maps the boot config's `cache.redis` block onto `RedisConfig`, reading the TLS files, and releases it with the other components. @@ -130,7 +131,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `config/` — Configuration -- **config.go** — Loads *boot* configuration from a YAML file with environment variable overrides (using [cleanenv](https://github.com/ilyakaznacheev/cleanenv)); every key has a `WH_`-prefixed env var. Boot config is only what can't change under a running process — the implementation each layer runs on, the process's `roles`, resource sizing, listeners, observability exporters, the settings-directory path, and the secrets (`clickhouse.password`, `cache.redis.password`, `auth.jwt_secret`, `auth.operator_key`). Everything tenant-tunable lives in the settings directory (`settings/`). Both sources are strict: `Load` refuses to boot naming every YAML key the struct doesn't declare (`strict.go`) and every `WH_*` environment variable no field binds (`check.go`), so a tunable that moved to the settings directory can't be read, ignored, and believed. Boot is the validator for this half — there is no dry-run command. See [Configuration Reference](/configuration). +- **config.go** — Loads *boot* configuration from a YAML file with environment variable overrides (using [cleanenv](https://github.com/ilyakaznacheev/cleanenv)); every key has a `WH_`-prefixed env var. Boot config is only what can't change under a running process — the implementation each layer runs on, the process's `roles`, resource sizing, listeners, observability exporters, the settings-directory path, the chtypes artifact directory, and the secrets (`clickhouse.password`, `cache.redis.password`, `auth.jwt_secret`, `auth.operator_key`). Everything tenant-tunable lives in the settings directory (`settings/`). Both sources are strict: `Load` refuses to boot naming every YAML key the struct doesn't declare (`strict.go`) and every `WH_*` environment variable no field binds (`check.go`), so a tunable that moved to the settings directory can't be read, ignored, and believed. Boot is the validator for this half — there is no dry-run command. See [Configuration Reference](/configuration). - **check.go** — `rejectUnboundEnv` is the environment half of the strict loader: `unboundEnv` walks the struct's `env` tags (plus the two process-level names, `WH_CONFIG` and `WH_LOG_LEVEL`) against the environment; `CheckDataDir` probes `data_dir` — run by `main` right after `Load` when `Config.NeedsDataDir` says a selected backend keeps state there, so an unusable `data_dir` refuses boot before anything dials out. It refuses an empty value (reachable through `WH_DATA_DIR=`) outright rather than probing the working directory; a path that exists and is not a directory; a dangling symlink at `data_dir` or any component above it (the walk to the nearest existing ancestor uses `Lstat`, so a failed mount is not skipped over as "does not exist"); and a directory the process cannot write to — or, when it does not exist, an unwritable nearest ancestor — probed by creating and removing one temp file. A permission denial, on the probe or on reaching the path through a parent without search permission, carries the UID-65532 hint, since a bind mount owned by root is the typical cause. - **backends.go** — the `.backend` keys: one string type per layer (`MQBackend`, `CacheBackend`, `DedupeBackend`, `CoordBackend`), each with its list of the backends this build has, and a `validate` per layer block (`checkBackend`, run by `validateBackends` in `Validate`, between `validateRoles` and `validateTopology`), which refuses a value not on the list and names the ones that are. One rule spans two layers: while `mq.backend` is `embedded`, `dedupe.lease` plus its own ceiling to the next whole second (`ceilSecond`) plus one more second must fit the embedded MQ's 2m duplicate window (`embeddedDuplicateWindow`), a cap of 59s (`maxEmbeddedLease`), because a client obeying the in-flight `503`'s `Retry-After` can republish as late as the lease plus that ceiling plus a second after the claim, since DynamoDB rounds a claim's expiry up to the second. A backend's own settings go in a `.` sub-block that is its `validate`'s case to check. `Distributed` reports whether the MQ is shared with other processes, `NeedsDataDir` whether a selected backend keeps state under `data_dir`, and `Warnings` returns what a valid configuration is still likely to get wrong — the combinations that are harmless or correct for one replica only (a shared MQ over a local cache or Pebble dedupe; under `nats`, `mq.max_bytes_gb` not applied; an `mq.nats` or `coord.nats` block its layer's backend ignores), a `cache.redis` block that is not read, certificate verification turned off — which `app.New` logs at `WARN`. `mq.backend` has two values, `embedded` and `nats` (`MQNATS`), and `nats` reads the `mq.nats` sub-block, which `MQ.validate` checks only when it is selected. `coord.backend` takes `nats` (`CoordNATS`) too, whose `coord.nats` block holds only the bucket name: the leases ride `mq.nats`'s connection. - **cache_redis.go** — `CacheRedisConfig`, the `cache.redis` sub-block, and its checks: an address (exactly one in `standalone` mode, which dials only the first), each `host:port` with a port from 1 to 65535 (a URL or `user:password@` form refused without repeating it, since it may hold a password), a known mode (`sentinel` is refused until [#656](https://github.com/Wave-RF/WaveHouse/issues/656)), `db` 0 in cluster mode, positive timeouts and sizes, a `timeout` and `dial_timeout` of at most 1 s each (boot and `Close` each wait out a dial: a connect and a handshake bounded by `dial_timeout`, and a cluster's topology read bounded by the larger of the two), a `version_ttl` of at least 2 s, and a `compress_min_bytes` that is not negative (`0` never compresses); its defaults are in `defaults()` with the rest. `CacheRedisTLS.Config` builds the `tls.Config`, reading the files; `Validate` calls it so an unreadable file refuses boot, and `internal/app` calls it again to build the connection. A TLS key set while `tls.enabled` is off is an error rather than a plaintext connection. @@ -163,17 +164,17 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `typelayer/` — In-Process ClickHouse Parser -The only package that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` wraps one lazily-opened `chtypes.Registry`, found through a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`; empty means the chtypes search path), and only a process running the `api` role builds it — an ingest-worker-only or sweeper-only process never loads the artifact and boots without one. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for what ships where and how large it is. +The only package (with its test helper `typelayertest`) that imports `github.com/wave-rf/chtypes/go/chtypes`. One process-wide `Engine` opens one `chtypes.Registry` at boot (manifests only; an `api` process with no artifact anywhere on the search path refuses to boot), found through a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`; empty means the chtypes search path), and each ClickHouse line's library is opened lazily by the first tenant bound to it. Only a process running the `api` role builds it — an ingest-worker-only or sweeper-only process never loads the artifact and boots without one. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for what ships where and how large it is. Each tenant has its own table set inside that engine, bound from the tenant's own schema refresh and released when the tenant is no longer served. - **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind. A tenant whose ClickHouse line has no installed artifact is unavailable on its own, and so is one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Either way the cause is recorded (an ingest client sees only a generic `503`; the cause goes to the log), every other tenant keeps working, and a later bind of the same tenant (a reload, a refresh that now agrees) clears it. - **`Engine.RoleTable(tenant, table, shape)`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column, including one the role may not otherwise write — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column stays in the schema as `MATERIALIZED` of its default, so naming it is ClickHouse's code 117 while expressions that read it still compile, and an absent check column takes the claim as its default while a supplied value is accepted only if it equals the claim. A role-shape pool starts at one handle and grows to `min(GOMAXPROCS, 4)` when every handle is busy; a base table's grows to `min(GOMAXPROCS, 8)`. - **`Table.IngestWith(format, opts, body, checks...)`** (`Ingest` is the same with no options) runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. -- **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per table. Only a definite true admits; a predicate error, a policy column the table no longer has, schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. +- **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per table. Only a definite true admits; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (`decline`), schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. - A tenant that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row of that tenant's tables from a role that has a row `filter`, with reason `unavailable`. -- A role whose schema cannot be compiled — or that may write no column of the table — is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`, and the cause is logged once a minute; it is not the `503` an unavailable tenant gets, since retrying cannot fix a policy or a table. -- **`typelayertest`** (`internal/typelayer/typelayertest`) holds the test helpers other packages' tests use — `TestEngine`, `SkipWithoutArtifact`, `RequireEnv` — which only test binaries link (`typelayer` itself never imports `testing`). The `typelayer` package's own tests use a small in-package copy, since they cannot import a package that imports them. +- A role whose schema cannot be compiled — or that may write no column of the table — is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`, and the cause is logged once a minute; it is not the `503` an unavailable tenant gets, since retrying cannot fix a policy or a table. A shape that fails only because an injected check value cannot be its column's default (`'abc'` on a `UInt64`) is retried without the defaults and served: a record omitting that column then fails the check (`403` on an integer column, `422` on another), and only a shape that does not compile even without them gets the `500`. +- **`typelayertest`** (`internal/typelayer/typelayertest`) holds the test helpers other packages' tests use — `TestEngine`, `SkipWithoutArtifact`, and `RequireEnv` (the `WAVEHOUSE_TEST_REQUIRE_CHTYPES` switch that turns a skip into a failure) — which only test binaries link (`typelayer` itself never imports `testing`). The `typelayer` package's own tests use a small in-package copy, since they cannot import a package that imports them. See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest error-response shape and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced) for how predicates are compiled and evaluated. @@ -212,8 +213,8 @@ The package's design invariants — stdout always 100%, WARN+ERROR always export ### `policy/` — Access Control -- **policy.go** — Hasura-style policy types, now **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant carries a separate `SelectPermissions` and `InsertPermissions` — so a field only one side ever honored (`filter`, aggregations and the `max_*` limits on select; `check` on insert) does not exist on the other, and a document that puts one there fails the strict decode as an unknown key. `Evaluate()` resolves permissions against JWT claims (including `{{ jwt.claim.path }}` template resolution) for ONE operation, leaving the side it did not resolve **nil**. That is what the pointers buy: an *empty* side means "no restrictions" — what the admin return constructs on both sides — while a *nil* side means "you asked the wrong operation", and as value types the two were the same zero value. Every accessor fails closed on a nil side. The handful of bare field reads outside this package each sit past an accessor that denies an unresolved side first, so a nil `Select` never reaches one; if that ordering ever changed they would panic rather than silently widen. A nil guard that skips such a read must never be added, since an absent `WhereClause` is an unfiltered query. The per-column decision `IsColumnAllowed(col, insert)` takes the side it is being asked about, alongside its batch/projection forms `AllowedProjection()` and `RestrictsColumns()`, `IsAggregationAllowed()`, `CheckClauses()` — the write-side accessor for the one consumer that iterates a side's map instead of asking about a column, whose `ok=false` a caller must treat as *refuse the write*, never as *no checks to run* — `resolvePredicates()` (exposed through `Predicates()`) — the one resolution both read surfaces render from, so the SQL `WHERE` and the chtypes filter the stream compiles cannot drift — and `Validate()`, split into `validateSelectPerms`/`validateInsertPerms` and run from `settings.Validate` on every adoption, which is where the rules in [Access Control](/access-control) are actually enforced. -- **canonical.go** — the one rendering layer for policy comparison operands: every JWT claim value (`CanonicalScalar`) is rendered into one exact canonical decimal form (positional, digit-bounded, never a float64 round-trip) before it reaches a `filter`/`check` predicate, so every comparison surface binds the same value the same way, and a null/object/array claim fails closed. A policy-authored literal is *not* re-rendered — it binds exactly as written, and a spelling the column cannot read is ClickHouse's own type error at evaluation time on both surfaces. On an integer column the bound claim is additionally compared through the strict round-trip cast (`WhereSQL(colType)` renders the predicate for the query builder; the stream's chtypes filter uses the same shape). +- **policy.go** — Hasura-style policy types, now **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant carries a separate `SelectPermissions` and `InsertPermissions` — so a field only one side ever honored (`filter`, aggregations and the `max_*` limits on select; `check` on insert) does not exist on the other, and a document that puts one there fails the strict decode as an unknown key. `Evaluate()` resolves permissions against JWT claims (including `{{ jwt.claim.path }}` template resolution) for ONE operation, leaving the side it did not resolve **nil**. That is what the pointers buy: an *empty* side means "no restrictions" — what the admin return constructs on both sides — while a *nil* side means "you asked the wrong operation", and as value types the two were the same zero value. Every accessor fails closed on a nil side. The handful of bare field reads outside this package each sit past an accessor that denies an unresolved side first, so a nil `Select` never reaches one; if that ordering ever changed they would panic rather than silently widen. A nil guard that skips such a read must never be added, since a skipped `WhereSQL` is an unfiltered query. The per-column decision `IsColumnAllowed(col, insert)` takes the side it is being asked about, alongside its batch/projection forms `AllowedProjection()` and `RestrictsColumns()`, `IsAggregationAllowed()`, `CheckClauses()` — the write-side accessor for the one consumer that iterates a side's map instead of asking about a column, whose `ok=false` a caller must treat as *refuse the write*, never as *no checks to run* — `resolvePredicates()` (exposed through `Predicates()`) — the one resolution both read surfaces render from, so the SQL `WHERE` and the chtypes filter the stream compiles cannot drift — and `Validate()`, split into `validateSelectPerms`/`validateInsertPerms` and run from `settings.Validate` on every adoption, which is where the rules in [Access Control](/access-control) are actually enforced. +- **canonical.go** — the one rendering layer for policy comparison operands: a numeric JWT claim (`CanonicalScalar`) is rendered into one exact canonical decimal form (positional, digit-bounded, never a float64 round-trip) before it reaches a `filter`/`check` predicate, a string claim binds as is, and a null/object/array claim fails closed, so every comparison surface binds the same value the same way. A policy-authored literal is *not* re-rendered — it binds exactly as written. On an integer column every bound value, literal or claim, is compared through the strict round-trip cast (`ResolvedSelect.WhereSQL` in policy.go renders the predicate for the query builder; the stream's chtypes filter uses the same shape), so a spelling the column cannot hold is NULL and matches nothing. On any other column a spelling the type cannot read is ClickHouse's own error at evaluation time — code 53 on a `Decimal`, 72 on a `Float` — which fails the whole `/v1/query` read with `400 clickhouse.rejected` and withholds the row on the stream (reason `error`). - **source.go** — `Source`, a `func() *Policy` the `/v1/ops` gate (over a flat settings directory) reads per call, so a settings reload applies to the very next request; in production it is the default tenant's `settings.Store.Policy`, and `Static(p)` fixes one for tests. The tenant-aware surfaces take a keyed variant that resolves to the same `Store.Policy`: `api.PolicySource` (`func(*settings.Store) *policy.Policy`) for ingest, structured query and pipes, `stream.PolicySource` (`func(tenant.ID) *policy.Policy`) for the hub, and `auth.PolicySource` (same shape) for the operator key's admin role. A `nil` result is a deliberate lockout. Predicate *evaluation* lives elsewhere: `internal/typelayer` compiles a role's resolved predicates through chtypes and evaluates them — `Table.ParseRow` / `Row.Visible` for a streamed event, `Table.IngestWith` (a row filter inside the validating parse) for an ingest `check` — with ClickHouse's own comparison semantics for every column type. `policy` only resolves the values both the SQL path and that engine bind. @@ -251,7 +252,7 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi ### `chsql/` — ClickHouse SQL Helpers -- **chsql.go** — Dependency-free ClickHouse SQL helpers shared by `query/` and `policy/`, kept in their own package to break an import cycle. `QuoteIdent` is the single place every identifier — column, table, alias — becomes SQL text: always backtick-quoted and escaped, so any ClickHouse-legal name (dots, spaces, unicode, keywords) is safe. `BindUnsafe` reports whether a name contains a literal `?`, which would desync the positional-to-named parameter rewrite; such names are rejected fail-closed rather than silently mis-bound. `IntegerType` and `StrictInt` pick and render the strict round-trip cast a policy claim is compared through on an integer column, identically for the query builder and the type layer, so a claim that does not fit the column's type is NULL and matches nothing instead of wrapping. +- **chsql.go** — Dependency-free ClickHouse SQL helpers shared by `query/`, `policy/`, `typelayer/` and the ingest worker, kept in their own package to break an import cycle. `QuoteIdent` is the single place every identifier — column, table, alias — becomes SQL text: always backtick-quoted and escaped, so any ClickHouse-legal name (dots, spaces, unicode, keywords) is safe. `BindUnsafe` reports whether a name contains a literal `?`, which would desync the positional-to-named parameter rewrite; such names are rejected fail-closed rather than silently mis-bound. `EscapeStringParam` is the one encoding of a `{p:String}` value, used by the structured-query binder (`query.Bind`) and the type layer's filters alike. `IntegerType` and `StrictInt` pick and render the strict round-trip cast a policy claim is compared through on an integer column, identically for the query builder and the type layer, so a claim that does not fit the column's type is NULL and matches nothing instead of wrapping. ### `keyenc/` — Key Escaping @@ -286,13 +287,16 @@ Client POST /v1/ingest?table={table} error code (exception_code) and message — an unknown field, a MATERIALIZED/ALIAS column, or a column this role may not write is 117 (an EPHEMERAL value is accepted only where the format names columns, the - role may write it, a DEFAULT reads it and no computed column does; it feeds that DEFAULT, never stored or published); a record chtypes cannot answer for is a distinct "declined" outcome - (422), never a data rejection + role may write it, a DEFAULT reads it and no computed column does; it feeds that DEFAULT, never stored or published); a record chtypes did not answer for is a distinct "declined" outcome + (422) — the shape, or a body it could not read as a whole, which + declines every record in it → Evaluate the role's check clauses over the accepted rows with one compiled chtypes filter — false is 403 for that record, unevaluable is 422 - → Optional dedupe: resolve the id (configurable ID field; a row missing it or - setting it to null is published un-deduped + logged/counted, or rejected - under require_id) — deliberately after validation, so a chtypes-rejected + → Optional dedupe: resolve the id from the row ClickHouse rendered + (configurable ID field; a cell that is absent, null or an empty string — + what an omitted String or Nullable id produces — is published un-deduped + + logged/counted, or rejected under require_id; an omitted id on a + numeric column renders as 0 and is deduped under it) — deliberately after validation, so a chtypes-rejected record is never marked seen → The steps below run per window of up to 256 accepted records → Reserve the window's (tenant, table, id) keys in one call: a duplicate is @@ -304,7 +308,8 @@ Client POST /v1/ingest?table={table} → Commit the published ids in one call; on a failed publish, commit the records before it and release the rest (a publish whose outcome is unknown keeps its claim until the lease lapses) - → 200 OK returned immediately, per-record outcomes in the response body + → 200 OK returned immediately with per-record outcomes (batch); a single + object answers with its record's own status → (If the tenant's NATS stream is full, or not open: 503 + Retry-After header, the id released) @@ -391,7 +396,7 @@ Client POST /v1/ops/query (browser, CDN, corp proxy) caches the result. ``` -The proxy-pattern wins are: zero classification logic on the WaveHouse side (no `IsMutation` heuristic to maintain), and any ClickHouse statement type — including verbs added in future versions and inline FORMAT overrides — works without WaveHouse code changes. Multi-statement input (`SELECT 1; TRUNCATE t`) is supported when the upstream ClickHouse has multi-query enabled, which is the default on recent versions; older or restrictively-configured servers will return a clear error from ClickHouse itself for the second statement. The proxy buffers the response in memory with a 64 MiB cap (502 with `clickhouse response exceeded N bytes` on overflow, to keep a runaway `SELECT *` from pinning RAM on the API server), and passes ClickHouse's `Content-Type` through when an inline `FORMAT` directive overrides the default JSON envelope. The structured query endpoint and pipes reach ClickHouse the same way, over its HTTP interface, but ask for `default_format=JSONEachRow` and bind their values as named `{pN:String}` parameters. ClickHouse renders the JSON; WaveHouse frames the lines into an array and caches those bytes, so a value is spelled by the connected server with no WaveHouse conversion table in between. The native driver now serves only schema discovery and the readiness ping. +The proxy-pattern wins are: the proxy itself classifies nothing (only pipes run `IsMutation`, to keep writes out of the cache), and any single ClickHouse statement — including verbs added in future versions and inline FORMAT overrides — works without WaveHouse code changes. ClickHouse's HTTP interface takes one statement per request: multi-statement input (`SELECT 1; TRUNCATE t`) is refused with code 62 (`Multi-statements are not allowed`), a `400 clickhouse.rejected`. The proxy buffers the response in memory with a 64 MiB cap (502 with `clickhouse response exceeded N bytes` on overflow, to keep a runaway `SELECT *` from pinning RAM on the API server), and passes ClickHouse's `Content-Type` through when an inline `FORMAT` directive overrides the default JSON envelope. The structured query endpoint and pipes reach ClickHouse the same way, over its HTTP interface, but ask for `default_format=JSONEachRow` and bind their values as named `{pN:String}` parameters. ClickHouse renders the JSON; WaveHouse frames the lines into an array and caches those bytes, so a value is spelled by the connected server with no WaveHouse conversion table in between. The native driver now serves only schema discovery and the readiness ping. ### Streaming Path @@ -436,7 +441,8 @@ Client GET /v1/stream → Policy filtering (historical + live): denied tables skipped, denied columns stripped, row filter compiled and evaluated per subscriber against claims by internal/typelayer (fail closed on a compile/parse - error, a missing policy column, schema drift, or an unavailable engine). + error, a missing policy column, a filter column the published row does + not carry, schema drift, or an unavailable engine). Column projection runs once per role (Hub.Broadcast) — per-subscriber work only where a row filter makes visibility per-connection; replay shares the same column policy + row check but projects per-connection diff --git a/docs/src/content/docs/claude-code.md b/docs/src/content/docs/claude-code.md index e2b3d2e5..aadc482a 100644 --- a/docs/src/content/docs/claude-code.md +++ b/docs/src/content/docs/claude-code.md @@ -15,7 +15,7 @@ If you're new to Claude Code itself, the [official docs](https://code.claude.com 2. **Authenticate**: log in to your Max subscription — Claude Code prompts you on first run. -3. **Bootstrap the repo**: `make tools`. This installs Go tools, pnpm deps, and **also configures git hooks** (`git config core.hooksPath .githooks`). Without this step, the team's pre-commit / pre-push gates won't fire. +3. **Bootstrap the repo**: `make tools`. This installs Go tools, pnpm deps, and **also configures git hooks** (`git config core.hooksPath .githooks`). Without this step, the team's pre-commit / pre-push gates won't fire. Then run `scripts/fetch-chtypes.sh` once per machine: the chtypes artifact the API process and the test suites need, which `make tools` does not fetch. 4. **Optional — worktrunk**: install [worktrunk](https://worktrunk.dev) for parallel-agent worktree management. The team's project hooks live in `.config/wt.toml`. @@ -139,7 +139,7 @@ User-specific worktrunk config goes in `~/.config/worktrunk/config.toml`; the co **We use `gh` CLI as the canonical GitHub access path**, not a GitHub MCP server. Reasons: -- `gh` is already a hard dev requirement +- Most contributors already use `gh` - Works identically in Claude Code, terminal, and shell scripts - No extra auth / approval / npx cold-start diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index b189d545..b721abc7 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -9,7 +9,7 @@ sidebar: import { Tabs, TabItem } from "@astrojs/starlight/components"; -WaveHouse is configured via a YAML file with environment variable overrides. All environment variables use the `WH_` prefix. +WaveHouse is configured via a YAML file with environment variable overrides. All WaveHouse environment variables use the `WH_` prefix (the OpenTelemetry SDK's `OTEL_*` and the chtypes SDK's `CHTYPES_*` are read by those libraries). ## Loading Order @@ -137,7 +137,7 @@ By default one process does all the work. `roles` splits it, so that the API and | Role | Runs | | --- | --- | -| `api` | The HTTP API, and what answers it: schema discovery, the token verifiers and their JWKS refresh, the dedupe stores, and the SSE hub with its bridge off the queue and its keepalive wheel. Every API process runs its own set of these, and each API process receives every event for its own SSE clients. | +| `api` | The HTTP API, and what answers it: the chtypes type layer (ingest validation and row filters), schema discovery, the token verifiers and their JWKS refresh, the dedupe stores, and the SSE hub with its bridge off the queue and its keepalive wheel. Every API process runs its own set of these, and each API process receives every event for its own SSE clients. | | `ingest` | The ingest worker, which writes the queue to ClickHouse. Under `mq.backend=nats` the ingest processes share the queue's shards out through membership leases in `coord.nats.bucket`, one process per shard at a time, so a table's rows are written by one process. | | `sweeper` | The sweeper, which purges messages that are written and older than their tenant's gap window. It runs under the `sweeper` lease, on `mq.backend=embedded` only. Under `mq.backend=nats` your streams' retention does its work, so it is not wired there, and a process whose only role is `sweeper` is refused (below). A tenant whose gap window is longer than the history stream keeps is checked by the API processes instead, at boot and after each reload, with one warning per tenant and window. | @@ -168,7 +168,7 @@ Only the secret, the connection ceiling and the chtypes artifact directory are b | --- | --- | ------- | ----------- | | `clickhouse.password` | `WH_CH_PASSWORD` | *(empty)* | Authentication password, combined with the settings directory's `clickhouse.username` on every (re)connect. A secret, so it never lives in a tracked JSON file; rotating it is a restart. | | `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the `max_open_conns` of the open pools — one per distinct connection tuple among the served tenants, see [the settings directory](/settings-directory#clickhouse) — must not add up to more. Pools above it at boot refuse to start, naming the sum and the ceiling; on a reload a pool resized above it is refused and keeps its size, and a new pool that would cross it is not opened — its tenants keep the pool they had, or have none — both logged (the reload itself still reports `adopted`) and retried by the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. It counts native connections only: the HTTP reader behind structured queries and pipes is capped separately, one cap per connection tuple (URL, user, password, database, TLS), at the largest `max_open_conns` among the tenants sharing it. | -| `clickhouse.chtypes_registry` | `WH_CHTYPES_REGISTRY` | *(empty)* | Directory holding the chtypes artifacts (one `/` per ClickHouse line). Empty defers to the chtypes search path — `$CHTYPES_REGISTRY`, `~/.cache/chtypes/artifacts/abi6/-`, then the system directories. Either way a library is opened lazily, on first use of its line; an explicit directory is searched first, then the rest of the path. Boot-tier: changing it is a restart. See [chtypes artifacts](/deployment#chtypes-artifacts). | +| `clickhouse.chtypes_registry` | `WH_CHTYPES_REGISTRY` | *(empty)* | Directory holding the chtypes artifacts (one `/` per ClickHouse line). Empty defers to the chtypes search path — `$CHTYPES_REGISTRY`, `~/.cache/chtypes/artifacts/abi6/-`, then the system directories. Either way a library is opened lazily, on first use of its line; an explicit directory is searched first, then the rest of the path. A directory set here that does not exist or cannot be read refuses boot in a process with the `api` role. Boot-tier: changing it is a restart. See [chtypes artifacts](/deployment#chtypes-artifacts). | ### ClickHouse user access @@ -177,7 +177,7 @@ The ClickHouse user WaveHouse connects as (the settings directory's `clickhouse. - **`SELECT` on the tenant's tables.** Structured queries and read pipes read through it. - **`INSERT` on the tenant's tables, for the ingest worker.** The worker writes each batch with `INSERT … FORMAT JSONCompactEachRow` over the HTTP interface. A process that only serves reads never inserts, but the same user is used by every process, so grant it wherever any process ingests. - **Read access to `system.columns` and `system.tables`**, plus `SELECT timezone()` and `SELECT version()`, which need no grant. Schema discovery reads the columns, their `DEFAULT`/`MATERIALIZED` kinds and each table's DDL, and records the server's version and time zone; the version picks the [chtypes artifact](/deployment#chtypes-artifacts) and the zone is the one the in-process parser reads bare timestamps in. ClickHouse lists only the tables a user holds a grant on, so a table the user cannot see is a table WaveHouse does not know. -- **A profile that lets it change per-query settings.** WaveHouse sends settings with every query: reads carry `readonly=2` together with `max_execution_time`, `wait_end_of_query`, `http_write_exception_in_output_format`, `cancel_http_readonly_queries_on_client_close` and the JSON/date rendering settings, and the ingest worker's `INSERT` carries `date_time_input_format`, `input_format_null_as_default` and `async_insert=0`. A profile with `readonly=1` (or a `` entry that pins any of these) makes ClickHouse refuse them: a read refused with `READONLY` answers `502 clickhouse.misconfigured`, a `` entry that a setting violates is refused as `400 clickhouse.rejected`, and the worker's batches fail. Use `readonly=0`, or `readonly=2` for a user that only ever reads: `readonly=2` still forbids the ingest worker's `INSERT`. +- **A profile that lets it change per-query settings.** WaveHouse sends settings with every query: reads carry `readonly=2` together with `max_execution_time`, `wait_end_of_query`, `http_write_exception_in_output_format`, `cancel_http_readonly_queries_on_client_close` and the JSON/date rendering settings, and the ingest worker's `INSERT` carries `date_time_input_format`, `input_format_null_as_default` and `async_insert=0`. A profile with `readonly=1` (or a `` entry that pins any of these) makes ClickHouse refuse them: a read refused with `READONLY` answers `502 clickhouse.misconfigured`, a `` entry that a setting violates is refused as `400 clickhouse.rejected`, and the worker's inserts are refused too: under `readonly=1` or `readonly=2` (code 164) its rows stay queued and are retried until the tenant's queue budget fills, and under a constraint that pins one of its settings (code 452) every row fails its row-by-row retry and is parked on the DLQ. Use `readonly=0`, or `readonly=2` for a user that only ever reads: `readonly=2` still forbids the ingest worker's `INSERT`. - **Whatever else the statements you hand it need.** A [pipe that writes](/pipes#pipes-that-write) and the admin-only raw SQL endpoint `/v1/ops/query` run as this user too, so a pipe that runs `ALTER` needs the grant for it, and a missing grant is `403 clickhouse.access_denied`. ### Server-side resource limits @@ -219,7 +219,7 @@ Set the backstop on the profile of the ClickHouse user WaveHouse connects as (th What a caller sees when one of these trips: a row or byte limit is `400 clickhouse.limit_exceeded`; a server-wide time, memory or quota limit is `503 clickhouse.unavailable` (retryable, so the SDK retries it — except on a [pipe that writes](/pipes#pipes-that-write), which is never retryable), since WaveHouse cannot tell it from server pressure — except on `/v1/query` for a role that sets its own `max_memory_usage`, where every memory-limit error is read as that cap and answered `400 clickhouse.limit_exceeded`. See [ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths). :::caution[How the two layers compose] -WaveHouse's per-role caps are sent as per-query `SETTINGS` on its connection, so they **compose** with the ClickHouse profile — a per-role cap *tightens* within the profile's ceiling, and a `` block bounds how far any setting can move. But if the profile marks a setting `readonly` (or `` disallows changing it), ClickHouse will **reject** WaveHouse's per-query override and the query fails. So keep the settings WaveHouse manages (`max_memory_usage`, `max_execution_time`, `max_rows_to_read`, `max_result_rows`) **changeable** for its user — use a `` constraint, not `readonly`, if you want a hard ceiling. +WaveHouse's per-role caps are sent as per-query `SETTINGS` on its connection, so they **compose** with the ClickHouse profile — a per-role cap *tightens* within the profile's ceiling, and a `` block bounds how far any setting can move. But if the profile marks a setting `readonly` (or `` disallows changing it), ClickHouse will **reject** WaveHouse's per-query override and the query fails. So keep every setting WaveHouse sends **changeable** for its user — the per-role caps (`max_memory_usage`, `max_execution_time`, `max_rows_to_read`, `max_result_rows`) and the fixed read and insert settings listed under [ClickHouse user access](#clickhouse-user-access) — and use a `` constraint, not `readonly`, if you want a hard ceiling. ::: ### Message Queue (NATS) diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 8a5196d1..d57dad99 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -11,7 +11,7 @@ How to run WaveHouse in production — one binary, Docker images, releases, heal ## One binary, plus a per-version artifact -WaveHouse runs as one process with embedded NATS and optional Pebble dedup. At start an API-role process also loads a second artifact — a per-ClickHouse-version shared library that runs ClickHouse's own parser in-process for ingest validation and row-level security (via [chtypes](#chtypes-artifacts)); a process that runs only the ingest worker or the sweeper loads none. The only external network dependency is ClickHouse, unless you select a shared backend: [`mq.backend: nats`](#external-nats) (and `coord.backend: nats`), [`cache.backend: redis`](#multiple-instances-and-the-shared-cache) or [`dedupe.backend: dynamodb`](#a-shared-dedupe-table-on-dynamodb). +WaveHouse runs as one process with embedded NATS and optional Pebble dedup. An API-role process also needs a second artifact — a per-ClickHouse-version shared library that runs ClickHouse's own parser in-process for ingest validation and row-level security (via [chtypes](#chtypes-artifacts)), found at start and opened when the first tenant on its ClickHouse line is bound; a process that runs only the ingest worker or the sweeper needs none. The only external network dependency is ClickHouse, unless you select a shared backend: [`mq.backend: nats`](#external-nats) (and `coord.backend: nats`), [`cache.backend: redis`](#multiple-instances-and-the-shared-cache) or [`dedupe.backend: dynamodb`](#a-shared-dedupe-table-on-dynamodb). ### Quick Start with Docker Compose @@ -53,6 +53,10 @@ This starts: # permissive "public" trial policy — replace policies.json before production make settings/config.json +# Once per machine: the chtypes artifact for ClickHouse 26.8, which an +# API-role process refuses to boot without (another line: see chtypes artifacts) +scripts/fetch-chtypes.sh + # Build make build @@ -88,7 +92,7 @@ Production images are published to GitHub Container Registry by the release work ghcr.io/wave-rf/wavehouse: ``` -`:vX.Y.Z` is the immutable per-release tag. A **stable** release also moves `:latest`; a **prerelease** moves `:alpha` / `:beta` / `:rc` / `:next` instead — chosen from the *first* prerelease identifier matched exactly, so `v0.2.0-rc.1` gives `:rc` while `-alpha1` or `-preview.1` give `:next` — one rule (`scripts/ci/release-channel.sh`), shared with the npm dist-tags — so a release candidate never displaces the `:latest` a shipped stable release owns. `:dev` is the rolling `main`-branch build, and `:dev-` (immutable, pruned after 30 days) captures a single commit. To pin (see the [alpha-stage caution](https://github.com/Wave-RF/WaveHouse#-project-status) in the README), use a `:dev-` tag — the full 40-character commit SHA, not the short form — or an image digest. +`:vX.Y.Z` is the immutable per-release tag. A **stable** release also moves `:latest`; a **prerelease** moves `:alpha` / `:beta` / `:rc` / `:next` instead — chosen from the *first* prerelease identifier matched exactly, so `v0.2.0-rc.1` gives `:rc` while `-alpha1` or `-preview.1` give `:next` — one rule (`scripts/ci/release-channel.sh`), shared with the npm dist-tags — so a release candidate never displaces the `:latest` a shipped stable release owns. `:dev` is the rolling `main`-branch build, and `:dev-` (immutable, pruned after 30 days) captures a single commit. To pin (see the [alpha-stage caution](https://github.com/Wave-RF/WaveHouse#project-status) in the README), use a `:dev-` tag — the full 40-character commit SHA, not the short form — or an image digest. Published images carry a signed [Sigstore](https://www.sigstore.dev/) build-provenance attestation (stored in the registry). Verify one before deploying, pinning the signer to the workflow that publishes the tag — `--repo` alone accepts an attestation from any workflow in the repo: @@ -112,13 +116,13 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ **Which processes load it.** Only processes with the `api` role — the ones that serve ingest and the stream. A process that runs only the ingest worker or the sweeper loads no artifact and boots without one installed. An API process refuses to start when no artifact is installed at all. -**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with the cause in the server log; run such tenants in a separate process, or align the servers' `timezone` setting. See [API → Ingest error responses](/api#error-responses) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). +**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with the cause in the server log; run such tenants in a separate process, or align the servers' `timezone` setting. The health probes stay green through either — `/livez` and `/readyz` do not read the type layer — so watch for the `ERROR` log line `chtypes cannot serve this tenant`, the ingest `503`s, and `wavehouse_sse_rows_withheld_total{reason="unavailable"}`. See [API → Ingest error responses](/api#post-v1ingesttabletable--ingest-data) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). -**Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse does not autofetch on a miss in production — an unmatched line is a boot-time or refresh-time failure, not a background download. +**Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse never fetches an artifact itself: a line with no installed artifact makes the tenants on it unavailable at their schema refresh, and boot is refused only when no artifact is installed at all, or when an explicit `clickhouse.chtypes_registry` does not exist or cannot be read. One exception: the SDK honors `CHTYPES_AUTOFETCH=1` from the environment over WaveHouse's setting, and would then download a missing line (hundreds of MB) inside a schema refresh — leave it unset. -**Size.** Each artifact is roughly 160–290 MB on disk; a running process holding several loaded versions (e.g. across a rolling ClickHouse upgrade) costs roughly 120 MB of resident memory per loaded version (the chtypes multi-version guide's figure; a library is opened on first use of its line, not at registry construction). +**Size.** Each artifact is roughly 160–300 MB on disk; a running process holding several loaded versions (e.g. across a rolling ClickHouse upgrade) costs roughly 120 MB of resident memory per loaded version (the chtypes multi-version guide's figure; a library is opened on first use of its line, not at registry construction). -**Docker images** ship the artifact(s) baked in: the image build fetches whatever `chtypes.lock` names (see below), so a container never needs network access to chtypes' artifact store at runtime. The image sets `CHTYPES_REGISTRY=/opt/chtypes/artifacts` (the SDK's own variable); set `WH_CHTYPES_REGISTRY` only to point at a bind-mounted directory instead. **The published images support ClickHouse 26.8 only**: they bake the lines `chtypes.lock` names, and a server on any other line answers every tenant on it `503` (with row-filtered streams withholding their rows) behind a generic body. For another line, fetch its artifact (below), mount the directory into the container and set `WH_CHTYPES_REGISTRY` to the mount path. +**Docker images** ship the artifact(s) baked in: the image build fetches the lines `scripts/fetch-chtypes.sh` lists (`LOCK_LINES`), each pinned by `chtypes.lock` (see below), so a container never needs network access to chtypes' artifact store at runtime. The image sets `CHTYPES_REGISTRY=/opt/chtypes/artifacts` (the SDK's own variable); set `WH_CHTYPES_REGISTRY` only to point at a bind-mounted directory instead. **The published images support ClickHouse 26.8 only**: they bake only that line, and a server on any other line answers every tenant on it `503` (with row-filtered streams withholding their rows) behind a generic body. For another line, fetch it for the container's platform into a directory of its own — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --platform linux-amd64 --dest ./chtypes-artifacts ` (`linux-arm64` on arm) — make it readable by the image's user (`chmod -R a+rX ./chtypes-artifacts`; the fetch writes each line's directory with mode `0700`), mount it read-only, and set `WH_CHTYPES_REGISTRY` to the mount path. That directory is searched first and the baked line stays available; a path that does not exist or cannot be read refuses boot. **Release archives and `go install` / building from source** do not carry or fetch an artifact — only the Docker images bake one in. See the [README's `go install` caveat](https://github.com/Wave-RF/WaveHouse#c-go-install-binary-no-docker). Fetch one yourself before first run. Both routes below run the chtypes command with `go run`, which needs Go 1.27 and a C compiler; a release archive has neither `scripts/fetch-chtypes.sh` nor the lock file, so use the second form there, or fetch from a checkout and copy the directory to the host: @@ -134,9 +138,9 @@ go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch `, once per platform (`darwin-arm64`, `linux-amd64`, `linux-arm64`), without `--frozen` — and commit the result; don't regenerate it implicitly. +`chtypes.lock`, checked in at the repo root, records the exact artifact file and SHA-256 per platform and ClickHouse patch (`linux-amd64/26.8.15.10-lts`) the project builds and tests against. CI restores from it with `--frozen` (refusing anything the lock doesn't name) rather than fetching the rolling artifact release, so a pipeline never silently starts testing a new build. Refresh it deliberately — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --lock chtypes.lock --platform `, once per platform (`darwin-arm64`, `linux-amd64`, `linux-arm64`), without `--frozen` — and commit the result; don't regenerate it implicitly. -A lock is specific to the SDK's ABI revision (6 at v0.5.2): the fetcher never selects a build from another revision, so after an SDK bump that changes the revision, `--frozen` fails (`CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED`) until the lock is regenerated the same way, and the CI cache key and path (`abi6`) move with it. +A lock is specific to the SDK's ABI revision (6 at v0.5.2): the fetcher never selects a build from another revision, so after an SDK bump that changes the revision, `--frozen` fails (`CHTYPES_ARTIFACT_PINNED` or `CHTYPES_ARTIFACT_UNPUBLISHED`) until the lock is regenerated the same way, and the `abi6` CI cache key and path in `.github/actions/setup-env/action.yml` must be bumped with it. ## Releases @@ -362,7 +366,7 @@ Until `startupProbe` succeeds, kubelet doesn't run `livenessProbe` or `readiness `SIGTERM` or `SIGINT` begins a graceful stop in three bounded phases whose budgets add up: 1. **Drain**, within [`server.shutdown_timeout`](/configuration#server) (default 10s). The listener stops accepting, every open [SSE stream](/api#get-v1stream--server-sent-events-stream) is ended at once, gap-fill in progress included (clients reconnect and resume from `Last-Event-ID`), and in-flight requests and the ingest worker's in-hand batches finish. A settings reload caught mid-hook gives up too. Whatever is still open at the deadline is force-closed. -2. **Release**, within a fixed 5s. The stores (embedded NATS, the Pebble dedupe store, the cache, ClickHouse) close; one still closing at the deadline is abandoned, the ones after it are left to the exit, and both are named in the log. +2. **Release**, within a fixed 5s. The stores (embedded NATS, the Pebble dedupe store, the cache, ClickHouse) and the type layer's compiled handles close; one still closing at the deadline is abandoned, the ones after it are left to the exit, and both are named in the log. 3. **Flush**, within a fixed 3s. Telemetry is flushed last, on its own budget, so the lines the release logged reach the collector even when a close was slow. Only the drain scales with the deployment's workload, so it is the one operators tune; the other two are constants. @@ -609,7 +613,7 @@ The folder name is the tenant id, and each folder is a complete settings directo **The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` takes it too, and reads a rejected or removed tenant's dead-letter queue like a served one's, since the queue is kept; under the embedded broker a tenant that has none is a `404`. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). -**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant and table), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it, and in each table. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Under the embedded broker its message queue is its own as well (under [`mq.backend: nats`](#limits-that-differ-from-the-embedded-queue) tenants share the partitions): its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history that gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a rejected row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached results (`POST /v1/query` and pipe alike) dropped the moment it is back on one, so a repaired or restored folder never serves rows cached before the inserts it missed. A tenant whose folder moves it to another address or database has them dropped too, since they came from other tables. Apart from those two drops, a cached pipe result stays until its TTL expires, since a pipe names no table and no insert invalidates it. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. +**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant and table), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it, and in each table. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Tenants on one ClickHouse line must share a server time zone to be served by one process: the first tenant bound fixes the line's zone, and a tenant whose server reports another is refused (`503`; see [chtypes artifacts](#chtypes-artifacts)). Under the embedded broker its message queue is its own as well (under [`mq.backend: nats`](#limits-that-differ-from-the-embedded-queue) tenants share the partitions): its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history that gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a rejected row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached results (`POST /v1/query` and pipe alike) dropped the moment it is back on one, so a repaired or restored folder never serves rows cached before the inserts it missed. A tenant whose folder moves it to another address or database has them dropped too, since they came from other tables. Apart from those two drops, a cached pipe result stays until its TTL expires, since a pipe names no table and no insert invalidates it. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. **What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. Tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its seen ids kept for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its open `GET /v1/stream` connections are ended, as any tenant's are when it stops being served — the other tenants' events are untouched. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. @@ -645,7 +649,7 @@ For local development, `docker compose -f deployments/compose/dependencies.yaml WaveHouse uses a **Bring Your Own Schema** model. You create your tables in ClickHouse with whatever columns and engines you need. WaveHouse discovers the schemas automatically via `system.columns` and validates ingest data against them — see [Schema Validation](/api#post-v1ingesttabletable--ingest-data) for the rules a record must satisfy. -Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's default where none is declared (`NULL` on a `Nullable` column), evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [the journey of one event](/ingest-pipeline#the-journey-of-one-event) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). +Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's default where none is declared (`NULL` on a `Nullable` column), evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [Ingest Pipeline → High-level shape](/ingest-pipeline#high-level-shape) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). Example table: @@ -756,8 +760,9 @@ The streaming surface loses something too, more quietly. SSE gap-fill (`?since=` **The upgrade does not carry the old queue over at all.** Boot deletes the earlier build's queue and dead-letter queue (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) and everything in them, logging a `WARN` with each one's message count: an event the old build had not yet inserted, and a row it had already parked, do not survive the upgrade. Draining first keeps the events not yet inserted; a row already parked is lost with the queue, since the earlier build offers no way to read one back (`GET /v1/ops/dlq/stats` returns counts only). -Two audits belong **before** the drain, because none of them announces itself afterwards: +Three audits belong **before** the drain, because none of them is cheap to discover afterwards: +- **API processes now need a chtypes artifact for their ClickHouse line.** The published images bake 26.8 only: a tenant whose server is on another line answers every ingest `503` until you mount that line's artifact ([chtypes artifacts](#chtypes-artifacts)), and a release-archive or `go install` deployment refuses to boot until one is fetched. Windows, FreeBSD and darwin/amd64 builds are no longer published. - **Policy `check` blocks are now validated against the table.** A `check` naming a column the table lacks, one it computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is a per-record `403` on *every* insert by that role. `wavehouse validate` cannot catch it — it never sees the ClickHouse schema — so audit them against their tables first. See [Access control → Insert checks](/access-control#insert-checks). - **Every `WH_*` variable the binary does not bind refuses boot.** The old binary ignored a variable it did not read; the new one names every unbound one and exits before it opens the queue, so a pod spec or compose file that still carries one comes back from the upgrade as a container that will not start. Diff the environment against the [Configuration Reference](/configuration) first: a `WH_*` variable that is not in its tables is unbound, and whatever it used to configure now lives in the [settings directory](/settings-directory) or is gone. A Kubernetes Service in the pod's namespace named `wh` or `wh-*` counts too: it injects link variables under the `WH_` prefix (`WH_SERVICE_HOST` and `WH_PORT` for `wh`, `WH_FOO_SERVICE_HOST` and `WH_FOO_PORT` for `wh-foo`), so set `enableServiceLinks: false` on the pod spec. diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index ef36edf0..8b541a7a 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -23,13 +23,13 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: ### The chtypes artifact — fetch it once per machine -`internal/typelayer` loads a per-ClickHouse-version shared library at start to run ingest validation and row-level security through ClickHouse's own parser (see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)). It is not source code and `make tools` does not fetch it for you — pull it once with: +`internal/typelayer` uses a per-ClickHouse-version shared library — looked for at start (an API process refuses to boot with none installed) and opened the first time a tenant on that ClickHouse line is bound — to run ingest validation and row-level security through ClickHouse's own parser (see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)). It is not source code and `make tools` does not fetch it for you — pull it once with: ```bash scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --frozen --lock chtypes.lock 26.8 ``` -It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–290 MB — expect the first run to take a minute or two. Without it the API process refuses to boot (`make dev`, `make test-e2e`, and the app that `make test-integration` and `make ci` start), and the unit tests that need the engine skip; set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (CI does) to make a missing artifact fail those tests instead. A `503` on ingest, with row-filtered stream rows withheld, is what a process that did find an artifact answers for a ClickHouse line the artifact does not cover. +It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–300 MB — expect the first run to take a minute or two. Without it the API process refuses to boot (`make dev`, `make test-e2e`, and the app that `make test-integration` and `make ci` start), and the unit tests that need the engine skip; set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (CI does) to make a missing artifact fail those tests instead. A `503` on ingest, with row-filtered stream rows withheld, is what a process that did find an artifact answers for a ClickHouse line the artifact does not cover. ### Auto-installed by `make tools` @@ -72,7 +72,7 @@ make tools scripts/fetch-chtypes.sh # once per machine (see above) # 2. Start ClickHouse (the only external dependency) -docker compose -f deployments/compose/dependencies.yaml up -d clickhouse +docker compose -f deployments/compose/dependencies.yaml up -d --wait clickhouse # 3. Create a table in ClickHouse docker compose -f deployments/compose/dependencies.yaml exec clickhouse \ @@ -102,7 +102,7 @@ WaveHouse is now running at `http://localhost:8080` in standalone mode with: On first run `make dev` seeds the gitignored `./settings` directory with `wavehouse bootstrap` and copies in the compose stack's trial policy (`deployments/compose/settings/policies.json` + `roles.json` — the `public` role: read/write `clicks`/`events`, no token). WaveHouse is otherwise fail-closed (the bootstrap seed ships no policy), so a `./settings` you emptied by hand denies every request until you put a policy back. Edits to any file in `./settings` hot-reload without a restart. -The tokenless data-plane calls work (create a `clicks` table first — see the [Getting Started](/getting-started) walkthrough): +The tokenless data-plane calls work (against the `clicks` table from step 3 of the [Quick Start](#quick-start) above): ```bash # Ingest an event @@ -167,13 +167,13 @@ These are the small targets behind `make dev` — useful directly when you want | `make deps-logs` | `docker compose logs -f clickhouse` (Ctrl+C detaches; container keeps running). | | `make deps-shell` | Drop into a `clickhouse-client` REPL on the running container. | | `make deps-wipe` | Stop ClickHouse **and destroy its data volume**. Use when you want a clean schema. | -| `make clean-all` | Nuclear option — every `make` artifact + dev/E2E containers + volumes + `data/`. | +| `make clean-all` | Nuclear option — every `make` artifact + the dev Compose containers and volumes + `data/`. | **Stopping `make dev`**: `Ctrl+C` stops air, which propagates SIGINT to WaveHouse for a graceful shutdown (NATS JetStream flush, etc.). ClickHouse stays up — re-running `make dev` is fast because the volume is preserved. Use `make deps-down` or `make deps-wipe` to stop ClickHouse explicitly. ### Running with observability -WaveHouse natively exports standard OpenTelemetry (OTLP) data to `127.0.0.1:4317`. Rather than coupling a heavy observability database stack to the dev server, we provide three lightweight, single-container dashboard options. +WaveHouse natively exports standard OpenTelemetry (OTLP) data — with no `OTEL_EXPORTER_OTLP_ENDPOINT` set, to `localhost:4317`, where these dashboards listen. Export is off by default (`otel.enabled: false` in `config.yaml`): the E2E fixture turns it on, so `make test-e2e` needs nothing, while for `make dev` set `otel.enabled: true` in `.config.local.yaml` or run `WH_OTEL_ENABLED=true make dev`. Rather than coupling a heavy observability database stack to the dev server, we provide three lightweight, single-container dashboard options. You run these in a separate terminal tab alongside `make dev` or your test suites (`make test-e2e`). @@ -188,7 +188,7 @@ They block the terminal and stream logs; simply press `Ctrl+C` to instantly tear **Typical Workflow:** 1. Open Tab 1: run `make obs-aspire` (UI opens automatically) -2. Open Tab 2: run `make dev` (or `make test-e2e`) +2. Open Tab 2: run `WH_OTEL_ENABLED=true make dev` (or `make test-e2e`) 3. View traces, metrics, and logs flowing into the UI instantly. No accounts or auth tokens required. ### Using the SDK against `make dev` @@ -370,7 +370,7 @@ Shared test utilities live in `internal/testutil/`. The packages log through `sl ### Adding New Tests - **Unit test for `internal/foo/`** → create `internal/foo/foo_test.go` (same package). -- **Integration test needing Docker** → add a subtest under `tests/integration/` (e.g. a new file with `//go:build integration`). A test of one package against its own external server — the shared cache backend against Redis, Valkey and Dragonfly containers — lives beside the package instead (`internal/cache/redis_integration_test.go`, same build tag), and the package is listed in the `test-integration` target. +- **Integration test needing Docker** → add a subtest under `tests/integration/` (e.g. a new file with `//go:build integration`). A test of one package against its own external server — the shared cache backend against Redis, Valkey and Dragonfly containers — lives beside the package instead (`internal/cache/redis_integration_test.go`, same build tag), and the package is listed in the `test-integration` target. In `internal/api` and `internal/mq` that target picks tagged tests by name (`-run '^Test(ExternalNATS|NewNATS|NATSPermissions_Refuse|Leases|Integration_)'`), so name a new one `TestIntegration_*` in `internal/api`, or extend the pattern. - **E2E test via SDK** → add a `tests/e2e/sdk/*.test.ts` file. These tests exercise the full pipeline (ingest → ClickHouse → query) through the TypeScript SDK. Run with `make test-e2e`. - **Test helpers** → add to `internal/testutil/` (Go) or `tests/e2e/sdk/helpers.ts` (E2E). @@ -397,14 +397,15 @@ The orchestrator always provisions its own stack — fresh ClickHouse and Redis ```bash docker compose -f deployments/compose/dependencies.yaml --profile redis up -d -WH_CONFIG=tests/e2e/fixtures/config.yaml WH_CACHE_REDIS_ADDRS=localhost:6379 go run ./cmd/wavehouse +mkdir -p tmp/e2e-settings && cp tests/e2e/fixtures/settings/*.json tmp/e2e-settings/ +WH_CONFIG=tests/e2e/fixtures/config.yaml WH_SETTINGS_DIR=tmp/e2e-settings WH_CACHE_REDIS_ADDRS=localhost:6379 go run ./cmd/wavehouse ``` -The fixture matters: the suite signs its tokens with its `sdk-dev-secret` and depends on its dedupe, DLQ, and 5s schema-refresh settings. Point the suite at a default `make dev` server (`jwt_secret: change-me-in-production`) and setup's schema calls are rejected, then global setup dies 30s later on a misleading `schema not refreshed within 30s`. The repo root matters too — the fixture's `settings.dir` is relative to the working directory. The fixture's settings directory (policy, pipes, and tunables) points at ClickHouse on `localhost:9000`; if yours isn't there, edit `clickhouse.addr` / `http_port` in `tests/e2e/fixtures/settings/config.json` (the orchestrator patches them itself for its testcontainer). +The suite writes policy and pipes into the server's settings directory, so give it a scratch copy, never the tracked fixture. The fixture matters: the suite signs its tokens with its `sdk-dev-secret` and depends on its dedupe, DLQ, and 5s schema-refresh settings. Point the suite at a default `make dev` server (`jwt_secret: change-me-in-production`) and setup's schema calls are rejected, then global setup dies 30s later on a misleading `schema not refreshed within 30s`. The repo root matters too — the relative paths above resolve against the working directory. The fixture's settings (policy, pipes, and tunables) point at ClickHouse on `localhost:9000`; if yours isn't there, edit `clickhouse.addr` / `http_port` in the copy's `config.json` (the orchestrator patches its own copy for its testcontainer). Prefixing the variable to `make dev` does **not** work: that recipe pins `WH_CONFIG=.config.local.yaml` inline, which overrides anything inherited from the environment. -Then set `CLICKHOUSE_URL` / `WAVEHOUSE_URL` and run `pnpm test` from `tests/e2e/sdk/`; teardown is a no-op on that path, so your stack survives between iterations. +Then, from `tests/e2e/sdk/`, run `CLICKHOUSE_URL=http://localhost:8123 WAVEHOUSE_URL=http://localhost:8080 WAVEHOUSE_SETTINGS_DIR=$(git rev-parse --show-toplevel)/tmp/e2e-settings pnpm test`; teardown is a no-op on that path, so your stack survives between iterations. If a previous run was killed (harness timeout, stop button, `SIGKILL`), it can leave a `wavehouse-cov` behind. That process shares `tmp/data` and `tmp/wavehouse-cov.log` with the next run and will corrupt it, so the orchestrator kills any leftover before starting and says so. @@ -475,11 +476,11 @@ WaveHouse/ │ ├── auth/ # JWT/JWKS authentication middleware │ ├── cache/ # Query cache: Ristretto L1 + the tenant-led version index; the Redis-compatible shared backend │ ├── chconn/ # ClickHouse pools, one per connection tuple (reconciled on settings reload) -│ ├── chsql/ # Shared ClickHouse SQL helpers (quoting + bind-safety) +│ ├── chsql/ # Shared ClickHouse SQL helpers (identifier quoting, bind-safety, {p:String} value encoding, the strict integer-claim cast) │ ├── config/ # YAML + env var configuration │ ├── coord/ # Leases with fencing tokens (in-process Local, RunElected, coordtest suite) │ ├── dedupe/ # Optional deduplication (Reserve/Commit/Release; Pebble or DynamoDB) -│ ├── discovery/ # ClickHouse schema introspection + validation +│ ├── discovery/ # ClickHouse schema introspection (system.columns/system.tables, server version + timezone) │ ├── ingest/ # Batch buffering + DLQ + Active Sweeper + shard claims │ ├── keyenc/ # One escaping for composite keys (NATS subject tokens, cache keys, dedupe keys) │ ├── mq/ # MQ boundary: the only NATS/JetStream importer @@ -490,20 +491,23 @@ WaveHouse/ │ ├── settings/ # Settings directory: validate, adopted snapshot, reload │ ├── stream/ # SSE fan-out: Hub, Subscriber queue, Bucket, keepalive wheel │ ├── tenant/ # Tenant id: type, grammar, reserved default, request header name -│ └── testutil/ # Shared test helpers and mocks (cachetest suite) +│ ├── testutil/ # Shared test helpers and mocks (cachetest suite) +│ └── typelayer/ # In-process ClickHouse parser (chtypes): ingest validation, insert checks, row-filter compilation ├── tests/ # Integration & E2E tests │ ├── integration/ # Go integration tests (//go:build integration) │ └── e2e/ # E2E suite (orchestrator + ClickHouse and Redis testcontainers) -│ ├── fixtures/ # ClickHouse DDL + config and settings-directory fixtures +│ ├── fixtures/ # Server config + settings-directory fixtures (tables come from sdk/tables.ts) │ └── sdk/ # E2E specs driven through the TypeScript SDK (Vitest) ├── clients/ # Client SDKs │ └── ts/ # TypeScript SDK (@wavehouse/sdk, pnpm workspace) ├── deployments/ -│ ├── compose/ # Docker Compose files (standalone.yaml, dependencies.yaml) +│ ├── compose/ # Docker Compose files (standalone.yaml, dependencies.yaml) + settings/ (trial settings directory) +│ ├── nats/ # External NATS JetStream topology (nack CRs + Helm values) │ ├── Dockerfile # Runtime image │ └── Dockerfile.goreleaser # Release image (built by GoReleaser) -├── scripts/ # E2E orchestrator, cov tool, CI/hook helpers +├── scripts/ # E2E orchestrator, cov tool, chtypes fetcher, CI/hook helpers ├── docs/ # Documentation +├── chtypes.lock # Pinned chtypes artifacts (scripts/fetch-chtypes.sh) ├── config.yaml # Default configuration file ├── Makefile # Build, test, lint, deploy targets ├── .golangci.yml # Linter configuration @@ -513,7 +517,7 @@ WaveHouse/ ## Code Conventions -- **Strict Go formatting**: Use `gofumpt` (a stricter superset of `gofmt`, enforced by CI). Run `make fmt` to format. +- **Strict Go formatting**: Use `gofumpt` (a stricter superset of `gofmt`, enforced by CI). `make fmt` checks it; `make fix` applies it. - **Interface-first design**: Core behaviors (`Cache`, `Deduplicator`, `Publisher`, `Subscriber`) are defined as interfaces so implementations can be swapped behind a stable contract. - **Package boundaries**: The `internal/` directory ensures packages are private to this module. - **Error handling**: Return errors to callers. Use `slog` for structured logging, through the default logger (`slog.InfoContext(ctx, …)` and its siblings) — constructors don't take a `*slog.Logger`; tests silence or capture it with `internal/testutil/logtest`. @@ -571,7 +575,7 @@ Run `make help` to see all targets. Key ones: | **Cleanup** (tiered — compose explicitly for partial resets) | | | `make clean` | Build outputs only (`bin/`, `dist/`, `clients/ts/dist/`, `docs/dist/`, `docs/.dev-dist/`) | | `make clean-test` | Test outputs only (`tmp/` — coverage data, logs, NATS state) | -| `make clean-tools` | Installed tools and pnpm deps (`.bin/`, `node_modules/`) | +| `make clean-tools` | Installed tools and the workspace members' pnpm deps (`.bin/`, `clients/ts`, `tests/e2e/sdk` and `docs` `node_modules/`) | | `make clean-all` | Full reset: above + `data/` + Docker volumes | All test targets accept `ARGS="..."` for pass-through `go test` flags. Build targets accept `TAGS="..."` for Go build tags. `V=1` switches to verbose `gotestsum` output. @@ -607,7 +611,7 @@ PRs are grouped per config to reduce noise. The npm config is pointed at the wor The GitHub Actions config names **two** directories. `directory: /` reaches `.github/workflows/` but does not descend into `.github/actions/*/action.yml`, so the `setup-env` composite action — which owns every cache in `ci.yml` — was invisible to Dependabot, and its pins went stale against upstream and diverged from `publish-npm.yml`, which Dependabot *does* track and which doesn't call `setup-env`. Listing its directory under `directories:` brings it into the same weekly group; **adding a composite action means adding its directory there**, because nothing else catches the drift. -`typescript` majors are held back (`ignore: version-update:semver-major`) because `tsup` vendors a `rollup-plugin-dts` that crashes on TypeScript 7 during `clients/ts`'s `prepare` script — i.e. inside `pnpm install`, which takes every Node job down at once. See the comment in `.github/dependabot.yml` for the condition that lets it be removed. +`typescript` majors are held back (`ignore: version-update:semver-major`) because `tsup` vendors a `rollup-plugin-dts` that crashes on TypeScript 7 during `clients/ts`'s `prepare` script — i.e. inside `pnpm install`, which takes every Node job down at once. See the comment in `.github/dependabot.yml` for the condition that lets it be removed. `eventsource-parser` majors are held back too: v4 drops the CJS build the SDK's `require` entry needs ([#492](https://github.com/Wave-RF/WaveHouse/issues/492)). **No auto-merge.** Dependabot PRs go through the same merge gate as any other PR — an approval from the `@Wave-RF/wavehouse-admins` team (the ruleset's `required_reviewers` rule) plus the required checks. (The former `dependabot-automerge.yml`, which auto-approved and merged patch/minor bumps hands-off, was removed — every bump now gets a human admin review.) diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index 0d68da22..1b31a3b7 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -110,7 +110,7 @@ A self-contained `wavehouse storage-check` preflight subcommand that bakes this If you see any of these, benchmark the `/nats` volume as above: -- `open dlq stream: ... context deadline exceeded`, or `open ingest stream: ...`, when a tenant's queue first opens, at the boot or reload that first serves the tenant. +- `open dlq stream: ... context deadline exceeded`, `dlq stream info: ...`, or `open ingest stream: ...`, when a tenant's queue first opens, at the boot or reload that first serves the tenant. - `mq queue not reconciled with settings; the next reload retries` with `join the queue: ... context deadline exceeded`, and that tenant's ingest answering `503`, when a tenant's queue opens while the server runs and the ingest worker's or the stream hub's consumer cannot join it in time. - Ingest p99 latency in the seconds, or occasional `200`s that take multiple seconds to return. - Intermittent `503 Service Unavailable` from `/v1/ingest` when ClickHouse is healthy (the worker can't drain fast enough because acking is `fsync`-bound). diff --git a/docs/src/content/docs/getting-started.md b/docs/src/content/docs/getting-started.md index 99f028e6..1e231737 100644 --- a/docs/src/content/docs/getting-started.md +++ b/docs/src/content/docs/getting-started.md @@ -5,7 +5,7 @@ sidebar: order: 2 --- -Run WaveHouse locally in under five minutes. WaveHouse ships as one binary plus the per-ClickHouse-version [chtypes artifact](/deployment#chtypes-artifacts) it loads at start, with ClickHouse as the only external network dependency; this walkthrough covers ingest, query, and real-time streaming. +Run WaveHouse locally in under five minutes. WaveHouse ships as one binary plus the per-ClickHouse-version [chtypes artifact](/deployment#chtypes-artifacts) it opens for your server's line, with ClickHouse as the only external network dependency; this walkthrough covers ingest, query, and real-time streaming. ## Prerequisites @@ -23,6 +23,8 @@ cd WaveHouse docker compose -f deployments/compose/standalone.yaml up -d ``` +The first `up` builds the WaveHouse image from source — a cgo compile plus a download of the pinned chtypes artifact, a few hundred MB — so expect several minutes once; later starts reuse the image. + The stack bind-mounts `deployments/compose/settings/` as WaveHouse's [settings directory](/settings-directory) — the hot-reloadable configuration, ClickHouse address included — so there is nothing to seed; edit those files and the running container picks the change up. This exposes: @@ -62,7 +64,7 @@ curl -s -X POST "http://localhost:8080/v1/ingest?table=clicks" \ # → {"ok":true} ``` -WaveHouse validates the body against the ClickHouse schema before acknowledging — using ClickHouse's own parser, running in-process (`internal/typelayer`, via [chtypes](/deployment#chtypes-artifacts)), so a rejection carries ClickHouse's own error code and message, the same as a native `INSERT` would produce. Unknown fields and type mismatches are rejected with a `400`. +WaveHouse validates the body against the ClickHouse schema before acknowledging — using ClickHouse's own parser, running in-process (`internal/typelayer`, via [chtypes](/deployment#chtypes-artifacts)), so a record ClickHouse refuses is rejected with a `400` carrying ClickHouse's own code and message: a value the column cannot read, such as `"score": "high"` (code 72), or a field the table lacks (code 117 — stricter than a native `INSERT`, which skips unknown fields by default). Values ClickHouse coerces are stored as a native `INSERT` would store them: `"42.5"` or `true` into `score` stores `42.5` or `1`. ## 4. Query @@ -93,7 +95,7 @@ curl -N "http://localhost:8080/v1/stream?table=clicks" curl -N "http://localhost:8080/v1/stream?table=clicks&since=2026-03-24T11:00:00Z" ``` -Rows arrive **positionally**, so raw `curl` output looks like `"row":["/home","signup"]` rather than named fields. The stream sends an `event: schema` frame before the first row and again when the column list changes, and a raw client must pair each row against the **most recent** frame rather than the first. +Rows arrive **positionally**, so raw `curl` output looks like `"row":["/home","signup",42.5,"2026-03-24T11:59:58.512Z"]` rather than named fields. The stream sends an `event: schema` frame before the first row and again when the column list changes, and a raw client must pair each row against the **most recent** frame rather than the first. That re-announcement is not guaranteed in one case: after a gap-fill across a column change, live rows can arrive without a fresh frame ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). Drop a row whose length disagrees with the last announced list rather than zipping it, and reconnect to resynchronize — an arity check cannot see a same-length change such as a `RENAME COLUMN`. See [the wire format](/api#get-v1stream--server-sent-events-stream) for the frame sequence and the full rule; the [TypeScript SDK](/sdk/streaming) does all of this for you. diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index d6c0d3bf..b9493940 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -63,7 +63,7 @@ The batch that reaches `insertToClickHouse` is not assembled from the request bo ::: :::note[Insert settings pinned] -Inserts pin the same parsing settings chtypes compiled the row with — `date_time_input_format=best_effort` and `input_format_null_as_default=1` — plus `async_insert=0`, which the worker adds itself (not a parsing setting, so it's never passed to chtypes): unpinned, ClickHouse 26.2+'s server-default async insert costs a flush-wait floor per statement and loses per-row error attribution (`While executing WaitForAsyncInsert` instead of `(at row N)`). `deduplicate_insert`'s server default (block-hash dedup, `enable` from 26.2 for plain `MergeTree`) is deliberately left unpinned — fine for retried-identical-batch at-least-once delivery, but worth knowing about if the worker's retry classifier ever needs to distinguish it from two legitimately identical batches. +Inserts pin the same parsing settings chtypes compiled the row with — `date_time_input_format=best_effort` and `input_format_null_as_default=1` — plus `async_insert=0`, which the worker adds itself (not a parsing setting, so it's never passed to chtypes): unpinned, ClickHouse 26.2+'s server-default async insert costs a flush-wait floor per statement and loses per-row error attribution (`While executing WaitForAsyncInsert` instead of `(at row N)`). `deduplicate_insert` (server default `enable` on 26.8) is deliberately left unpinned. It removes an identical retried block only on a table that keeps a deduplication log: a `Replicated*` table by default, a plain `MergeTree` only when `non_replicated_deduplication_window` is above its default of 0. It is worth knowing about if the worker's retry classifier ever needs to tell it apart from two legitimately identical batches. ::: ## The journey of one event diff --git a/docs/src/content/docs/pipes.mdx b/docs/src/content/docs/pipes.mdx index 0eeb0797..64ce45bd 100644 --- a/docs/src/content/docs/pipes.mdx +++ b/docs/src/content/docs/pipes.mdx @@ -78,7 +78,7 @@ Independently, every `required` declared parameter must be supplied regardless o ### How a value becomes SQL -Bound values are **inlined directly into the SQL string** (not sent as positional driver parameters — that lets a parameter sit anywhere ClickHouse allows a literal, including `LIMIT`). Inlining is type-aware and escaped, and a string brings its own quotes, so **write each placeholder bare, never inside quotes** — `WHERE id = {{id}}`, not `WHERE id = '{{id}}'`: +Bound values are **inlined directly into the SQL string** (not sent as ClickHouse query parameters — inlining lets a placeholder sit anywhere ClickHouse allows a literal, including `LIMIT`). Inlining is type-aware and escaped, and a string brings its own quotes, so **write each placeholder bare, never inside quotes** — `WHERE id = {{id}}`, not `WHERE id = '{{id}}'`: | Supplied value | Rendered as | Note | | -------------- | ----------- | ---- | @@ -169,12 +169,13 @@ curl -X POST http://localhost:8080/v1/pipes/top_pages \ -d '{"start_date": "2024-01-01", "limit": 20}' ``` -The response is a JSON array of rows. Results flow through the query cache (in-process by default, or a Redis shared by every instance with [`cache.backend: redis`](/configuration#cache)) with singleflight coalescing, so concurrent identical calls hit ClickHouse once; an `X-Cache: HIT` or `X-Cache: MISS` header tells you which path served the response. A [pipe that writes](#pipes-that-write) skips both and answers `X-Cache: BYPASS`. +The response is a JSON array of rows. Results flow through the query cache (in-process by default, or a Redis shared by every instance with [`cache.backend: redis`](/configuration#cache)) with singleflight coalescing, so concurrent identical calls hit ClickHouse once; an `X-Cache: HIT` or `X-Cache: MISS` header tells you which path served the response. A [pipe that writes](#pipes-that-write) skips both and answers `X-Cache: BYPASS`. Don't give a pipe's SQL its own `FORMAT` clause: WaveHouse asks ClickHouse for `JSONEachRow` and frames the lines into the response array, so any other format fails every call with `500 clickhouse.unknown`. | Status | Body | Cause | | ------ | ---- | ----- | | 404 | `{"error":"pipe not found"}` | No pipe registered under that name | -| 403 | `{"error":"forbidden"}` | Caller's role isn't in `allowed_roles` (and isn't admin) | +| 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid or expired token was supplied and the pipe's allowlist then denied the request | +| 403 | `{"error":"forbidden"}` (empty-role variant: `forbidden: request has no role and no public default_role is configured`) | Caller's role isn't in `allowed_roles` (and isn't admin) | | 400 | `{"error":"missing required parameter: x"}` | A required parameter wasn't supplied | | 400 | `{"error":"parameter \"x\": unsupported parameter type object"}` | A non-scalar value with no SQL form — a JSON object (directly, or nested in an array). An empty array is likewise rejected (`array parameter must not be empty`). | diff --git a/docs/src/content/docs/reverse-proxy.mdx b/docs/src/content/docs/reverse-proxy.mdx index 04331df8..405e69a4 100644 --- a/docs/src/content/docs/reverse-proxy.mdx +++ b/docs/src/content/docs/reverse-proxy.mdx @@ -93,7 +93,7 @@ WaveHouse caps the inbound request body it will decode, as a memory-safety backs | `POST /v1/query`, `GET/POST /v1/pipes/{name}` (parameter / AST bodies) | **1 MiB** | `413 {"error":"request body exceeded 1048576 bytes"}` | | `POST /v1/ingest`, `POST /v1/ops/query` (bulk payload bodies) | **16 MiB** | `413 {"error":"request body exceeded 16777216 bytes"}` | -These caps are **fixed and not configurable** — they aren't a tuning knob, they're an invariant. A JSON request body amplifies roughly an order of magnitude when decoded into memory (a large array or object of small values explodes into Go's in-memory representation), so an *uncapped* decoder on a public endpoint is a single-request out-of-memory vector. A query or pipe-parameter body is bounded by nature — a real one is far under 1 MiB even with a large `in`-list — so the 1 MiB cap is generous headroom for ordinary use. A structured query with a very large `in` list is the case that can reach it, since [the list travels in the body](/api#post-v1querytabletable--structured-query), so size such lists with the cap in mind. +These caps are **fixed and not configurable** — they aren't a tuning knob, they're an invariant. A structured-query or pipe-parameter body is decoded into Go values and amplifies roughly an order of magnitude in memory (a large array or object of small values explodes into Go's in-memory representation), so an *uncapped* decoder there is a single-request out-of-memory vector; an ingest or raw-SQL body is not decoded, and its cap bounds the buffered body (see the note below). A query or pipe-parameter body is bounded by nature — a real one is far under 1 MiB even with a large `in`-list — so the 1 MiB cap is generous headroom for ordinary use. A structured query with a very large `in` list is the case that can reach it, since [the list travels in the body](/api#post-v1querytabletable--structured-query), so size such lists with the cap in mind. Set your own **outer** limit at the proxy, sized to your real needs: @@ -201,7 +201,7 @@ These limit the *whole* request regardless of traffic, so no keepalive extends t WaveHouse does **not** derive a client IP from forwarded headers — it does no per-IP logic (rate limiting and IP allow/deny are the proxy's job) and does not trust `X-Forwarded-For` / `X-Real-IP` / `True-Client-IP` to rewrite the connection's source address. So a forged forwarded header has no effect on WaveHouse, and `r.RemoteAddr` (what OpenTelemetry records as the peer) is the honest immediate peer — your proxy, when one is in front. Still, don't expose `:8080` to untrusted clients: bind WaveHouse to a private interface or firewall the port so the proxy is the only path in. Capturing the real client IP in WaveHouse's own traces and logs — trusted-proxy-aware, so it can't be spoofed — is tracked in [#333](https://github.com/Wave-RF/WaveHouse/issues/333). ::: -- **`Content-Type`** — forward it **verbatim**; ingest reads the format from it and refuses anything it cannot read ([details](/api#post-v1ingesttabletable--ingest-data)). Appending rather than replacing is safe *only* if the proxy sends a second header **line** — those are resolved together and accepted when they agree. A proxy that **merges** duplicates into one comma-joined value (Envoy's `append: true`, and some WAF rewrite rules) produces `application/json, application/json`, which is a `415` on every ingest **even though both halves agree**. Replacing it is worse than merging, because it fails silently: an NDJSON batch declared `application/json` is read as the single object it starts with and the remaining lines are dropped behind a `200` ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)) — no error to alert on. And a proxy that rewrites `text/csv` or `text/tab-separated-values` (including their `; header=present` variants, which select the header-line formats) will turn a valid positional batch into a `415` or, if it drops the parameter, into a different reading of the header line (a rejected record under `header=absent`, a silently consumed header on a bare type). If ingest starts returning `415` fleet-wide after a proxy change, look here first. +- **`Content-Type`** — forward it **verbatim**; ingest reads the format from it and refuses anything it cannot read ([details](/api#post-v1ingesttabletable--ingest-data)). Appending rather than replacing is safe *only* if the proxy sends a second header **line** — those are resolved together and accepted when they agree. A proxy that **merges** duplicates into one comma-joined value (Envoy's `append: true`, and some WAF rewrite rules) produces `application/json, application/json`, which is a `415` on every ingest **even though both halves agree**. Replacing it is worse than merging, because it fails silently: an NDJSON batch declared `application/json` is read as the single object it starts with and the remaining lines are dropped behind a `200` ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)) — no error to alert on. And a proxy that rewrites `text/csv` or `text/tab-separated-values` (including their `; header=present` variants, which select the header-line formats) will turn a valid positional batch into a `415`, or change how its first line is read: rewritten to `header=absent`, a header line becomes a rejected record; with the parameter dropped, ClickHouse's auto-detection consumes it. If ingest starts returning `415` fleet-wide after a proxy change, look here first. - **CORS** — WaveHouse applies its own CORS from the settings directory's `cors.allowed_origins` — over a nested directory, [the list of the tenant the request names](/deployment#multi-tenant-deployments), the preflight following the `X-Tenant-ID` the proxy stamps on `OPTIONS`. Let one layer own CORS: either pass it through the proxy untouched (recommended), or strip it from WaveHouse and do it at the proxy — not both, or browsers see duplicate `Access-Control-Allow-Origin` headers and reject the response. diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index 5abe056d..7e8c49e9 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -33,6 +33,7 @@ The SDK **never throws** for anything the server returns — all API errors come | 401 | `HTTP_401` | No | On REST, a present-but-invalid or expired JWT that a gate then denied. **WaveHouse itself** never returns `401` for a *missing* token — that resolves to `default_role`, and a denial is `403`. On a stream it is always from something in front, since `/v1/stream` is ungated | | 403 | `HTTP_403` | No | Insufficient permissions | | 404 | `HTTP_404` | No | Table, pipe, or tenant not found | +| 422 | `HTTP_422` | No | Ingest only: the validation engine could not judge a record (`validation engine declined: …`), or an insert check could not be evaluated. It says neither that the data is bad nor that it was accepted; a single-object insert answers it as the call's error, a batch as a per-record `error` with no `exception_code`. The SDK does not retry it | | 400 | `clickhouse.rejected` / `clickhouse.limit_exceeded` | No | ClickHouse refused the query (bad SQL, an unknown column, a type mismatch) or it outran a limit — including the role's own caps | | 403 | `clickhouse.access_denied` | No | ClickHouse's user lacks a grant the statement needs | | 500 | `HTTP_500` | Yes, unless the body says `retryable: false` | Server error (retried per `maxRetries`) | @@ -172,7 +173,7 @@ export interface ClicksRow { | ClickHouse Type | TypeScript Type | |----------------|-----------------| | `String`, `FixedString`, `UUID`, `DateTime*`, `Date*`, `Enum*`, `IPv4/6` | `string` | -| `UInt*`, `Int*`, `Float*`, `Decimal*` | `number` — `Decimal*` comes back as a JSON number, not a string | +| `UInt*`, `Int*`, `Float*`, `Decimal*` | `number` — 64-bit and wider integers and `Decimal*` come back as JSON numbers, not strings, so `JSON.parse` rounds a value past 2^53 — store an id that large as a `String` column to keep every digit; a `Float*` NaN or infinity comes back as `null` | | `Bool` | `boolean` | | `Nullable(T)` | `T \| null` | | `Array(T)` | `T[]` | diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index e082225a..ace71a56 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -97,9 +97,9 @@ interface StreamEvent { } ``` -`data` is a row **object**, as it always has been — but the wire underneath is positional. The server sends the column list in its own `event: schema` frame — before the first row, and again whenever the list drifts on the **live** path (with one exception after a gap-fill, below) — and each row as a JSON array; the SDK keeps the announced list and zips every row against it, so this shape is unchanged and nothing in your code moves. It matters in two places. The row object has a **null prototype**: a ClickHouse column may legitimately be named `__proto__`, and on an ordinary object that assignment hits the inherited setter and the value disappears — so the SDK builds each row with `Object.create(null)`. Property access, spreading, `JSON.stringify` and destructuring all behave normally; what does not is anything inherited from `Object.prototype`, so use `Object.hasOwn(row, "x")` rather than `row.hasOwnProperty("x")`, and don't rely on `` `${row}` `` or `row.constructor`. (`liveQuery`'s REST backfill half still yields ordinary objects.) And a **raw** SSE consumer (a hand-rolled `EventSource`) must do the zipping itself — see [the wire format](/api#get-v1stream--server-sent-events-stream). A column the producer omitted is no longer always `null` on the wire: WaveHouse's ingest validation runs ClickHouse's own parser in-process, which evaluates the column's `DEFAULT` (or its type's default — `null` only for a `Nullable` column with none) before the row is published — the same as a native `INSERT` naming fewer columns than the table has. +`data` is a row **object**, as it always has been — but the wire underneath is positional. The server sends the column list in its own `event: schema` frame — before the first row, and again whenever the list drifts on the **live** path (with one exception after a gap-fill, below) — and each row as a JSON array; the SDK keeps the announced list and zips every row against it, so this shape is unchanged and nothing in your code moves. It matters in two places. The row object has a **null prototype**: a ClickHouse column may legitimately be named `__proto__`, and on an ordinary object that assignment hits the inherited setter and the value disappears — so the SDK builds each row with `Object.create(null)`. Property access, spreading, `JSON.stringify` and destructuring all behave normally; what does not is anything inherited from `Object.prototype`, so use `Object.hasOwn(row, "x")` rather than `row.hasOwnProperty("x")`, and don't rely on `` `${row}` `` or `row.constructor`. (`liveQuery`'s REST backfill half still yields ordinary objects.) And a **raw** SSE consumer (a hand-rolled `EventSource`) must do the zipping itself — see [the wire format](/api#get-v1stream--server-sent-events-stream). A column the producer omitted is no longer always `null` on the wire: WaveHouse's ingest validation runs ClickHouse's own parser in-process, which evaluates the column's `DEFAULT` (or its type's default — `null` only for a `Nullable` column with none) before the row is published — the same as a native `INSERT` naming fewer columns than the table has. One exception: a row inserted by a role that may write only some columns carries only those. The server announces that narrower list, and the other keys are absent from `data`, not `null`. -Row values of top-level `DateTime`/`DateTime64` columns inside `data` (not timestamps nested in `Array`/`Map`/`Tuple` columns) arrive as RFC 3339 in UTC, whatever zone the column declares — `"2026-06-21T04:00:00.123Z"` — matching what `/v1/query` returns for the same row byte-for-byte, by construction (see [Timestamp rendering](/api#timestamp-rendering)). `new Date(value)` parses it directly, and because the stream's `timestamp` field and a timestamp column use the same form, the SDK's comparisons between them (the live-query dedupe below, client-side `.where()` on a timestamp column) are by instant, not by spelling. A `DateTime64(3)` on a whole second arrives as `…:00.000Z`, and an event published before an upgrade, or by an older instance during a rolling deploy, replays in the spelling it was published in. +Row values of `DateTime`/`DateTime64` columns inside `data`, including those nested in an `Array`, `Map` or `Tuple`, arrive as RFC 3339 in UTC, whatever zone the column declares — `"2026-06-21T04:00:00.123Z"` — matching what `/v1/query` returns for the same row byte-for-byte, by construction (see [Timestamp rendering](/api#timestamp-rendering)). `new Date(value)` parses it directly, and because the stream's `timestamp` field and a timestamp column use the same zone and format (RFC 3339, UTC — though not the same precision: `timestamp` trims trailing zeros, a column keeps its scale), the SDK's comparisons between them (the live-query dedupe below, client-side `.where()` on a timestamp column) are by instant, not by spelling. A `DateTime64(3)` on a whole second arrives as `…:00.000Z`, and an event published before an upgrade, or by an older instance during a rolling deploy, replays in the spelling it was published in. ### Transport Behavior @@ -133,7 +133,7 @@ Streams go through `options.fetch`, `options.headers`, and `options.fetchOptions ### Server-Side Policy Filtering -Access-control policy applies on the server before anything reaches the client: tables the connection's role can't `select` are skipped, denied columns are stripped from each event, and the role's row-level `filter` is evaluated per subscriber against the connection's JWT claims. A stream on a row-policied table therefore delivers only the rows the policy admits for that connection — and, where the server's in-memory comparison can't prove a match, fewer; see [Access control — where each rule is enforced](/access-control#where-each-rule-is-enforced). Claims are captured when the connection opens: a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), while token expiry or claim changes take effect on reconnect. +Access-control policy applies on the server before anything reaches the client: tables the connection's role can't `select` are skipped, denied columns are stripped from each event, and the role's row-level `filter` is evaluated per subscriber against the connection's JWT claims. A stream on a row-policied table therefore delivers only the rows the policy admits for that connection — and, where the in-process ClickHouse evaluation does not answer a definite true (a predicate error, a filter on a column the inserting role could not write, an unbound tenant, schema drift), fewer; see [Access control — where each rule is enforced](/access-control#where-each-rule-is-enforced). Claims are captured when the connection opens: a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), while token expiry or claim changes take effect on reconnect. ### Client-Side Stream Filtering diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 38d01224..1d1f9d12 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -68,7 +68,7 @@ The access-control policy — one [policy document](/access-control#anatomy-of-a } ``` -An empty document (`{}`) means **no policy**: it validates with a warning, and the server adopts it fail-closed — every request is denied, including one carrying the admin role; only the [operator key](/access-control#operator-key) gets through. A non-empty document must pass the full [policy validation](/access-control#anatomy-of-a-policy) (column and row rules, the claim-template grammar, resource limits), and its roles must be declared in `roles.json`. The layout is **role-first** — `tables.
..select` — and a document still using the pre-v2 operation-first nesting is rejected, with the message linking straight to [the migration note](/access-control#migrating-from-the-operation-first-layout). A role named after an operation is the shape the two layouts most visibly collide in, and it gets its own message rather than the migration one: `tables.
.select.select` grants the same access under either reading and is accepted, but `tables.
.select.insert` means a read-only role named `insert` under the old layout and a write-only role named `select` under the new one — so validation refuses to guess and asks you to rename the role. Three states warn rather than fail: `default_role` equal to the admin role (every roleless request is admin — dev only), a grant keyed by the admin role (admin is an unconditional bypass, so the grant has no effect), and a grant that sets neither `select` nor `insert` (the role gets no access to the table — most often a half-finished migration). A leftover pre-v2 `"select": {}` block decodes into a grant keyed by a role literally named `select`. Whether you see this warning or an error then depends on `roles.json`: if `select` is not declared there — the usual case — the undeclared-role error fires first and you never reach the warning; if it *is* declared, you get the warning and the document adopts. +An empty document (`{}`) means **no policy**: it validates with a warning, and the server adopts it fail-closed — every request is denied, including one carrying the admin role; only the [operator key](/access-control#operator-key) still reaches `/v1/ops/*`. A non-empty document must pass the full [policy validation](/access-control#anatomy-of-a-policy) (column and row rules, the claim-template grammar, resource limits), and its roles must be declared in `roles.json`. The layout is **role-first** — `tables.
..select` — and a document still using the pre-v2 operation-first nesting is rejected, with the message linking straight to [the migration note](/access-control#migrating-from-the-operation-first-layout). A role named after an operation is the shape the two layouts most visibly collide in, and it gets its own message rather than the migration one: `tables.
.select.select` grants the same access under either reading and is accepted, but `tables.
.select.insert` means a read-only role named `insert` under the old layout and a write-only role named `select` under the new one — so validation refuses to guess and asks you to rename the role. Three states warn rather than fail: `default_role` equal to the admin role (every roleless request is admin — dev only), a grant keyed by the admin role (admin is an unconditional bypass, so the grant has no effect), and a grant that sets neither `select` nor `insert` (the role gets no access to the table — most often a half-finished migration). A leftover pre-v2 `"select": {}` block decodes into a grant keyed by a role literally named `select`. Whether you see this warning or an error then depends on `roles.json`: if `select` is not declared there — the usual case — the undeclared-role error fires first and you never reach the warning; if it *is* declared, you get the warning and the document adopts. A reload applies to the very next request: the policy is read per request off the adopted snapshot, and a live `GET /v1/stream` connection's role is re-evaluated against the new policy on the next event. @@ -127,8 +127,8 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `dedupe.tables.
.{id_field, require_id, retention}` | `{}` | Optional per-table overrides; each entry overrides only the fields it names and inherits the rest. | | `dlq.enabled` | `true` | Park poison rows — those ClickHouse still rejects after row-by-row isolation, and every row of a batch whose tenant has no ClickHouse connection — on the tenant's dead-letter stream (`DLQ_{tenant}`) (`false`: leave them unacked for redelivery — except an envelope the worker cannot read, which is dropped and counted) — see [Dead Letter Queue](#dead-letter-queue). | | `dlq.tables.
.enabled` | `{}` | Optional per-table override of the switch. | -| `query.timestamp_bucket_seconds` | `60` | Bucket (seconds, `>= 0`) that a structured query's relative time range is truncated to, so near-identical queries share a cache entry; `0` disables bucketing. Read per query. | -| `query.default_max_rows` | `10000` | Fallback result `LIMIT` (`>= 1`) applied to a structured query when the caller and policy specify none. A result-**shaping** default, not a resource limit — server-wide limits (memory, rows scanned, execution time) belong in ClickHouse, see [Server-side resource limits](/configuration#server-side-resource-limits). | +| `query.timestamp_bucket_seconds` | `60` | Bucket (seconds, `>= 0`) that both bounds of a structured query's `time_range`, relative or absolute, are truncated down to, so near-identical queries share a cache entry; `0` disables bucketing. Read per query. | +| `query.default_max_rows` | `10000` | Result `LIMIT` ceiling (`>= 1`) for a structured query: applied when the caller sends no `limit`, and clamps one above it. A role's `max_rows` can lower it, never raise it. A result-**shaping** default, not a resource limit — server-wide limits (memory, rows scanned, execution time) belong in ClickHouse, see [Server-side resource limits](/configuration#server-side-resource-limits). | | `schema.refresh_interval` | `60` | Seconds between ClickHouse table-schema re-discoveries (`>= 1`), per tenant; a reloaded value takes effect from the next refresh cycle, and the first periodic refresh lands at a random point within the interval. Schemas are also refreshable on demand via `POST /v1/ops/schema/refresh` (admin-only). | | `stream.keepalive_interval` | `30` | Seconds (`>= 1`) a quiet `GET /v1/stream` connection may go without a write before the server sends a `:` keepalive comment — keep it under your proxy's idle timeout; see [Streaming](#streaming). | | `stream.keepalive_buckets` | `3` | Load-spreading (`>= 1`): connections are spread across N buckets so each tick nudges ~1/N of live streams. Most deployments leave it. | @@ -188,14 +188,14 @@ What stays in boot config is only what cannot change under a running process — Every per-tenant dedupe knob lives here. Where the seen ids are kept (`dedupe.backend`) and how long a claim is held (`dedupe.lease`) are [boot config](/configuration#dedupe), the same for every tenant. The switch and its fields are resolved per record from one snapshot (table override → global value): - `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens or closes this tenant's store (in the embedded Pebble instance at `/pebble`, or its share of the DynamoDB table under `dedupe.backend: dynamodb`), so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`503 dedupe store unavailable`, `Retry-After: 5`) until it opens — the files asked for dedupe, so publishing un-deduped is not a fallback. With `dedupe.backend: pebble` that is the next reload or restart, and at boot a failed open over a flat directory refuses to start, like every other store; with `dynamodb` it is the background retry described below. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) with `dedupe.backend: pebble`, every tenant's seen ids live in that one instance, each key led by its tenant and table, and it is open while any tenant's switch is on: each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and keeps its seen ids for the folder that restores it; and if that instance fails to open, at boot or on reload, every tenant with dedupe on fails closed — its ingest answers `503 dedupe store unavailable` (`Retry-After: 5`) until a reload opens it — while the tenants with dedupe off carry on. Under `dedupe.backend: dynamodb` the table check plays the instance's part, in either shape: the table is checked whether or not any tenant's switch is on, and a table that fails it fails every tenant with dedupe on closed until the check, retried in the background and at once after every reload, passes. Only a misconfigured table (missing, the wrong key schema, access denied) over a flat directory whose tenant has dedupe on refuses boot instead ([Configuration](/configuration#dynamodb-dedupe)). -- `dedupe.id_field` (seed default `event_id`) — the **column** whose stored value is the dedup key. It is read out of the row ClickHouse rendered, not out of the request body, so the key is the value that will be stored rather than the caller's spelling (`256` into a `UInt8` keys on `0`) — identical for the documented case, a string id. "Missing" therefore means the row carries no value for it, which is what an omitted `String` column produces; a numeric id column cannot tell an omitted `0` from a supplied one. An id is a duplicate only within its own tenant and table: the same value in two tables is two ids. An id longer than 1,024 bytes once escaped (every byte but an ASCII letter, digit, `_` or `-` takes three) is stored as its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. While its record is being published, an id is held for its lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default): another request carrying the same id meanwhile gets `503` (`a request with the same dedupe id is in flight`) with the lease, in whole seconds, as `Retry-After` — see [the ingest errors](/api#post-v1ingesttabletable--ingest-data). An id is committed only after its record is published; if that commit fails (counted by `wavehouse_ingest_dedupe_commit_failed_total`, which should stay at zero), the record is still answered `ok` and the id lapses with its lease: a retry of it before then answers in-flight, one inside the ingest queue's two-minute duplicate window is dropped there by its idempotency key, and one after that is stored again. -- `dedupe.require_id` (seed default `false`) — controls what happens to a row with no value for `id_field` — omitted, or `null` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. +- `dedupe.id_field` (seed default `event_id`) — the **column** whose stored value is the dedup key. It is read out of the row ClickHouse rendered, not out of the request body, so the key is the value that will be stored rather than the caller's spelling (`256` into a `UInt8` keys on `0`) — identical for the documented case, a string id. "Missing" therefore means the row carries no value for it, which is what an omitted `String` column produces; a numeric id column cannot tell an omitted `0` from a supplied one. An id is a duplicate only within its own tenant and table: the same value in two tables is two ids. An id longer than 1,024 bytes once escaped (every byte but an ASCII letter, digit, `_` or `-` takes three) is stored as its SHA-256, counted by `wavehouse_dedupe_hashed_id_total`. While its record is being published, an id is held for its lease ([`dedupe.lease`](/configuration#dedupe), 30 seconds by default): another request carrying the same id meanwhile gets `503` (`a request with the same dedupe id is in flight`) with the lease, in whole seconds, as `Retry-After` — see [the ingest errors](/api#post-v1ingesttabletable--ingest-data). An id is committed only after its record is published; if that commit fails (counted by `wavehouse_ingest_dedupe_commit_failed_total`, which should stay at zero), the record is still answered `ok` and the id lapses with its lease: a retry of it before then answers in-flight, one inside the ingest queue's duplicate window (two minutes on the embedded broker) is dropped there by its idempotency key, and one after that is stored again. +- `dedupe.require_id` (seed default `false`) — controls what happens to a row with no value for `id_field` — a `null` or empty-string cell, which is what an omitted `String` or `Nullable` id column stores (it can't be deduped, so idempotency wouldn't apply to it); an omitted numeric id stores `0`, which is a value and is deduped under it. Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. - `dedupe.retention` (optional; seed default `"0"`) — how long a committed id stays a duplicate, as a Go duration string: `"24h"`, `"720h"` (30 days), `"90m"`. There is no day unit. `"0"` keeps every id forever, which was the only behavior before this key existed, and a `config.json` without the key means the same. Once an id's retention has ended, the next record carrying it is published as new. With `dedupe.backend: pebble`, a background sweep over the shared Pebble instance deletes the expired id: first about a minute after the instance opens (when the first tenant switches dedupe on), then hourly while any tenant keeps it on, counted by `wavehouse_dedupe_swept_keys_total{reason="expired"}`. With `dynamodb`, no sweep runs: the table's TTL on `ex` deletes the item ([Deployment](/deployment#a-shared-dedupe-table-on-dynamodb)). A finite retention must be at least `"2m"`, the embedded ingest queue's duplicate window (under `mq.backend: nats`, keep it at least the partitions' `duplicate_window` too, which boot warns about but this file cannot check; see [External NATS](/deployment#create-the-topology)): every deduped record is published under an idempotency key derived from its id, so an id re-sent after a shorter retention would be claimed again and then dropped by the queue as a copy, while the client was told it was accepted. A retention below that is refused, not raised to the minimum; so are a negative value and anything that is not a duration, such as `"30d"`, a number with no unit (`"300"` needs one: `"300s"`; `"0"` is the one exception), or a JSON number rather than a string. Hot-reloadable: a change applies to ids committed after the reload, and an id already committed keeps the expiry it was stored with. - `dedupe.tables.
.{id_field, require_id, retention}` — per-table overrides; each entry overrides only the fields it names and inherits the rest, so a table with no `retention` keeps the tenant's (forever when the tenant sets none). A table can keep ids for a shorter time than its tenant, or for longer, or forever (`"retention": "0"`) under a finite tenant retention. ## ClickHouse -The `clickhouse` block is the connection wiring, minus the password. A reload applies it to every consumer (schema discovery, structured queries, pipes, `/readyz`, the ingest worker's HTTP `INSERT`s, and the raw-SQL proxy — the HTTP-side ones re-read the target per request). A change to `addr`, `database`, `username` or the `tls` block moves the tenant to the pool of its new tuple, opened for it when no served tenant has that tuple; a change to `max_open_conns` or `max_idle_conns` resizes its pool; a change to `http_port`, `http_scheme`, `headers` or `query_timeout` needs no new connection. A replaced connection stays open for the longest `query_timeout` among the tenants that were on it, so in-flight queries finish. The change is **unconditional**: the adopted settings are the authority, so an address that isn't reachable is applied all the same and shows up where reachability already does — schema discovery retries and logs, `/readyz` fails, queries return errors — until the next reload fixes it. Two things are refused instead: a `tls` block whose certificate files cannot be loaded (unreadable, not PEM, or a `cert_file` and `key_file` that do not pair), and a pool that would put the process over the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) — a `max_open_conns` raised above it keeps the pool at its size, and a new connection tuple that would cross it is not opened. An option the ClickHouse driver refuses to open is refused the same way, though validation already excludes every one it knows of. A refused tenant keeps the pool and wiring it had until the next reload, or, newly served with no pool to keep, stays on none and its queries answer `503`. The reload itself still reports `adopted` — the refusal is an `ERROR` log line, not a finding — and the next reload retries it; at boot either refuses to start. Validation checks shape only (`host:port`, port range, scheme, non-empty database and user, timeout `>= 1`, the `tls` block's shape without opening its paths, header names and values, pool sizes); reachability is a runtime concern, so `wavehouse validate` needs no ClickHouse and no certificate files. At boot the address is dialed lazily, as before: an unreachable ClickHouse degrades `/livez` and retries rather than refusing to start. +The `clickhouse` block is the connection wiring, minus the password. A reload applies it to every consumer (schema discovery, structured queries, pipes, `/readyz`, the ingest worker's HTTP `INSERT`s, and the raw-SQL proxy — the HTTP-side ones re-read the target per request). A change to `addr`, `database`, `username` or the `tls` block moves the tenant to the pool of its new tuple, opened for it when no served tenant has that tuple; a change to `max_open_conns` or `max_idle_conns` resizes its pool; a change to `http_port`, `http_scheme`, `headers` or `query_timeout` needs no new connection. A replaced connection stays open for the longest `query_timeout` among the tenants that were on it, so in-flight schema discovery finishes (queries and pipes run over the HTTP interface, which the swap does not cut). The change is **unconditional**: the adopted settings are the authority, so an address that isn't reachable is applied all the same and shows up where reachability already does — schema discovery retries and logs, `/readyz` fails, queries return errors — until the next reload fixes it. Two things are refused instead: a `tls` block whose certificate files cannot be loaded (unreadable, not PEM, or a `cert_file` and `key_file` that do not pair), and a pool that would put the process over the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) — a `max_open_conns` raised above it keeps the pool at its size, and a new connection tuple that would cross it is not opened. An option the ClickHouse driver refuses to open is refused the same way, though validation already excludes every one it knows of. A refused tenant keeps the pool and wiring it had until the next reload, or, newly served with no pool to keep, stays on none and its queries answer `503`. The reload itself still reports `adopted` — the refusal is an `ERROR` log line, not a finding — and the next reload retries it; at boot either refuses to start. Validation checks shape only (`host:port`, port range, scheme, non-empty database and user, timeout `>= 1`, the `tls` block's shape without opening its paths, header names and values, pool sizes); reachability is a runtime concern, so `wavehouse validate` needs no ClickHouse and no certificate files. At boot the address is dialed lazily, as before: an unreachable ClickHouse degrades `/livez` and retries rather than refusing to start. **TLS.** There are two hops and two switches: `tls.enabled` puts the native-protocol connection (`addr`) on TLS, and `http_scheme: "https"` does the same for the HTTP interface (`http_port`). The rest of the `tls` block — the authority bundle, a client certificate, `insecure_skip_verify`, `server_name` — applies to whichever hop uses TLS, so a ClickHouse behind a private authority needs `ca_file` once for both. The certificate files are read when a pool opens: at boot, and on a reload that names a tuple no open pool has — a changed `tls` block, `addr`, `database` or `username`. A reload that changes only `http_port`, `http_scheme`, `headers`, `query_timeout` or the pool sizes reuses the material already loaded, so a file replaced in place is picked up the next time a pool for its tuple opens, or by a restart, like a rotated secret. Both switches move the hop to ClickHouse's TLS listeners, so the ports move too: `addr` to the secure native port (`9440` by default) and `http_port` to the HTTPS one (`8443`), as the example above does. Validation warns when only one hop is on TLS: a plaintext HTTP hop carries the credentials on every query and insert, and a plaintext native hop sends the password in its handshake. In the container images these are container paths: bind-mount the bundle (`-v /srv/clickhouse-ca.pem:/etc/wavehouse/clickhouse-ca.pem:ro`) and make it readable by UID 65532, like the settings directory. diff --git a/docs/src/content/docs/why-wavehouse.md b/docs/src/content/docs/why-wavehouse.md index 193f6f24..87a1ca71 100644 --- a/docs/src/content/docs/why-wavehouse.md +++ b/docs/src/content/docs/why-wavehouse.md @@ -49,11 +49,11 @@ The `parts_to_delay_insert` and `parts_to_throw_insert` thresholds are documente Even if you remember to batch client-side, a naive ingest path has no safe way to tell a client "slow down" or "that payload was malformed": -- **Validation happens late.** ClickHouse will accept an insert with a `String` where you expected a `UInt32` — it just rejects the whole block at parse time, and only after the network round-trip. There is no "this field is unknown" signal at the HTTP boundary; you build that yourself. +- **Validation happens late.** ClickHouse takes the request, then rejects the whole block at parse time when one row has a `String` where you expected a `UInt32` — after the network round-trip — and by default silently drops a field it does not know. There is no "this field is unknown" signal at the HTTP boundary; you build that yourself. - **No backpressure channel.** If the merger falls behind, ClickHouse raises an error at the *next* insert. The client has already left. - **No DLQ.** Bad events that fail to insert are either lost or logged into ClickHouse's error log. Good luck replaying yesterday's dropped rows. -WaveHouse fixes all three at the gateway: validates every payload against the real `system.columns` schema before accepting, returns `503 Service Unavailable` with a `Retry-After` header when the NATS WAL fills, and retries a ClickHouse outage with backoff while routing rows ClickHouse rejects to a dedicated dead-letter stream, one per tenant, you can inspect via `GET /v1/ops/dlq/stats`. +WaveHouse fixes all three at the gateway: validates every payload with ClickHouse's own parser, compiled from the real `system.columns` schema, before accepting, returns `503 Service Unavailable` with a `Retry-After` header when the NATS WAL fills, and retries a ClickHouse outage with backoff while routing rows ClickHouse rejects to a dead-letter stream (one per tenant on the embedded queue) you can inspect via `GET /v1/ops/dlq/stats`. ### No real-time push @@ -110,7 +110,7 @@ flowchart TB ### No row/column access control -ClickHouse has users and role grants, but nothing like row-level security driven by a JWT claim. If your product serves multiple tenants from a shared table, you're writing middleware to inject `WHERE tenant_id = ?` on every query — and hoping you never miss one. WaveHouse ships Hasura-style policies as a JSON file in the settings directory: per-role `allow_columns`, row-level `filter` with JWT claim templating (`{{ jwt.app_metadata.tenant_id }}`), validated on every load and hot-reloaded on edit. +ClickHouse has users and role grants, but nothing like row-level security driven by a JWT claim. If your product serves multiple tenants from a shared table, you're writing middleware to inject `WHERE tenant_id = ?` on every query — and hoping you never miss one. WaveHouse ships Hasura-style policies as a JSON file in the settings directory: per-role `allow_columns`, row-level `filter` with JWT claim templating (`{{ jwt.app_metadata.tenant_id }}`), validated on every load and hot-reloaded on edit (or on SIGHUP or the reload endpoint, for a nested directory). ## Part II — What people actually build instead @@ -156,7 +156,7 @@ flowchart TB | Real-time push | WebSocket service + bridge from Kafka | Built in (`/v1/stream`) | | Schema validation | Custom code in ingest API | Built in (discovers `system.columns`) | | Row/column access control | Custom middleware or a dedicated service | Built in (Hasura-style, JWT-driven) | -| Dead letter queue | Custom retry + dead topic on Kafka | Built in (a dead-letter stream per tenant) | +| Dead letter queue | Custom retry + dead topic on Kafka | Built in (a dead-letter stream per tenant on the embedded queue) | | Client SDK | Each team writes one | `@wavehouse/sdk` (TypeScript, one dependency, codegen) | The DIY path works — big teams run it — but the ops cost is not small. You're paying for a Kafka cluster (or Confluent bill), a second service you wrote from scratch, and all the debugging hours when the batching consumer stalls at 3 a.m. @@ -190,7 +190,7 @@ Tinybird wins on "zero ops to start." WaveHouse wins on "own your data plane and | Self-hosted | ✓ | ✓ | ✗ | ✓ | | Handles N-row inserts safely | ✗ merge blowup | ✓ via Kafka | ✓ | ✓ native | | Schema validation at the edge | ✗ | Custom | ✓ | ✓ (discovers schema) | -| Dead letter queue | ✗ | Custom | Partial | ✓ dead-letter stream per tenant | +| Dead letter queue | ✗ | Custom | Partial | ✓ dead-letter stream (per tenant on the embedded queue) | | Backpressure (503 + Retry-After) | ✗ | Custom | ✓ | ✓ | | Idempotent ingest (dedup by ID) | ✗ | Custom | ✓ | ✓ optional | | Real-time push (SSE) | ✗ | Custom service | ✗ | ✓ native, gap-fill | @@ -243,7 +243,7 @@ flowchart TB | NATS JetStream publish | ~1 ms | ~5 ms | | API `200 OK` to client | ~2 ms | ~8 ms | | Hub broadcast to SSE subscriber | ~1 ms | ~10 ms | -| Batch flush to ClickHouse | 5 s (configurable) | 5 s + ClickHouse insert time | +| Batch flush to ClickHouse | ≤ 5 s (sooner at 500 rows; fixed) | 5 s + ClickHouse insert time | | Query cache hit (L1) | < 0.5 ms | ~1 ms | | Query cache miss → ClickHouse | depends on query | depends on query | From 4950a598e5e60d2b7cf6c21d120450a78db2c31f Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 10:22:41 -0400 Subject: [PATCH 50/70] ci: the fetch script takes no --frozen flag; it always fetches frozen Co-Authored-By: Claude Opus 5.5 --- .github/actions/setup-env/action.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/actions/setup-env/action.yml b/.github/actions/setup-env/action.yml index f0fb45fa..0e3ce65a 100644 --- a/.github/actions/setup-env/action.yml +++ b/.github/actions/setup-env/action.yml @@ -211,8 +211,8 @@ runs: restore-keys: | chtypes-abi6-${{ runner.os }}-${{ runner.arch }}- - # Runs on both cache hit and miss: scripts/fetch-chtypes.sh --frozen is - # cheap on a hit (a manifest check against chtypes.lock, not a + # Runs on both cache hit and miss: scripts/fetch-chtypes.sh (always a + # --frozen fetch against chtypes.lock) is cheap on a hit (a manifest check against chtypes.lock, not a # re-download — see the script) and it is what turns a restored-but- # unverified cache entry back into a hash-checked one every run. - name: Fetch pinned chtypes artifact From 289d55aa9f66943d071ae97cd82696717cc4728d Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 10:41:16 -0400 Subject: [PATCH 51/70] docs: make dev OTel endpoint, operator-key scope, ClickHouse statuses - development.md: the OpenTelemetry SDK's unset default dials localhost:4317 over TLS, and the local dashboards speak plaintext OTLP, so make dev needs OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317 alongside WH_OTEL_ENABLED (the E2E fixture already sets it). - The operator key reaches /v1/ops/* always and the data plane only while a policy is adopted; with none, the nil policy denies its data-plane requests. configuration.mdx's table row, api.md, deployment.md's env comment and the standalone compose comment now say so. - architecture.md: ClickHouse's HTTP status is not a flat 500 (a syntax error is 400, an unknown table 404, TIMEOUT_EXCEEDED 408 on 26.8); it just does not say which kind of failure it was. - api.md: the stream matches the query path on a server running the line's default settings; a server profile can split the two. Co-Authored-By: Claude Opus 5.5 --- deployments/compose/standalone.yaml | 5 +++-- docs/src/content/docs/api.md | 6 +++--- docs/src/content/docs/architecture.md | 10 ++++++---- docs/src/content/docs/configuration.mdx | 2 +- docs/src/content/docs/deployment.md | 6 +++--- docs/src/content/docs/development.md | 4 ++-- 6 files changed, 18 insertions(+), 15 deletions(-) diff --git a/deployments/compose/standalone.yaml b/deployments/compose/standalone.yaml index 48820d21..8189ad60 100644 --- a/deployments/compose/standalone.yaml +++ b/deployments/compose/standalone.yaml @@ -40,8 +40,9 @@ services: # WH_AUTH_JWT_SECRET: "change-me" # # Optional non-JWT operator credential: a request presenting it - # (Authorization: Operator , or the X-Operator-Key alias) gets - # full-access admin, works even if the policy is wiped (break-glass). + # (Authorization: Operator , or the X-Operator-Key alias) reaches + # /v1/ops/* always, even with the policy wiped (break-glass), and the + # whole data plane while a policy is adopted. # Treat it as an admin secret. # WH_AUTH_OPERATOR_KEY: "change-me-operator-key" # diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index afb6753d..1f25f848 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -31,7 +31,7 @@ Prefer the header wherever you can. The TypeScript SDK streams over `fetch` and **Public (unauthenticated) access is driven by the policy.** Define a usable `default_role` and no-token requests are evaluated as that role (see [Roles & Access Control](#roles--access-control)); remove it and roleless requests are denied. Setting `default_role` equal to the `admin_role` is allowed — it makes every unauthenticated request admin (including `/v1/ops/*`), handy for local/dev — but it is logged loudly on every node that loads such a policy and must not be used in production. `/v1/ops/*` (raw SQL, pipe inspection, settings reload, schema, DLQ) is admin-only, and a pipe with **no `allowed_roles` authorizes nobody but the admin role** — but a pipe *can* be reached by the public when its `allowed_roles` lists the role the `default_role` resolves to (pipe access is plain allowlist membership, the same as any other role). -**Operator key (non-JWT, break-glass).** A separate, role-free credential — `auth.operator_key` — authorizes a caller as a **full-access platform operator**: the entire data plane *and* the `/v1/ops/*` surface, without a JWT and independently of the token verifier. Present it in the standard `Authorization` header with the `Operator` scheme (forwarded verbatim by proxies, no collision with Bearer JWTs), or via the `X-Operator-Key` alias: +**Operator key (non-JWT, break-glass).** A separate, role-free credential — `auth.operator_key` — authorizes a caller as a **full-access platform operator** — the `/v1/ops/*` surface always, and the entire data plane while a policy is adopted (with none, its data-plane requests are denied like anyone else's) — without a JWT and independently of the token verifier. Present it in the standard `Authorization` header with the `Operator` scheme (forwarded verbatim by proxies, no collision with Bearer JWTs), or via the `X-Operator-Key` alias: ```text Authorization: Operator @@ -39,7 +39,7 @@ Authorization: Operator X-Operator-Key: ``` -It is checked *before* the Bearer token (so it wins when both are present), compared in constant time, and — unlike a JWT bearing the `admin_role` — is honored **even when no policy is adopted** (an empty `policies.json`), making it the only credential that can still trigger `POST /v1/ops/settings/reload` over HTTP after the file is fixed. This deliberately bends "authentication is decoupled from authorization": a matching key both authenticates and authorizes in one step. It is disabled when empty (the default). See [Configuration — Authentication](/configuration#authentication) and [Access Control — Operator key](/access-control#operator-key). +It is checked *before* the Bearer token (so it wins when both are present), compared in constant time, and — unlike a JWT bearing the `admin_role` — is honored on `/v1/ops/*` **even when no policy is adopted** (an empty `policies.json`), making it the only credential that can still trigger `POST /v1/ops/settings/reload` over HTTP after the file is fixed. This deliberately bends "authentication is decoupled from authorization": a matching key both authenticates and authorizes in one step. It is disabled when empty (the default). See [Configuration — Authentication](/configuration#authentication) and [Access Control — Operator key](/access-control#operator-key). ### Roles & Access Control @@ -686,7 +686,7 @@ Each SSE connection is bound to a single `?table=`; to consume multiple tables, Values of `DateTime`/`DateTime64` columns inside `row`, nested ones included, are ClickHouse's own rendering of the stored value — the exact bytes chtypes' `RowsExport` produced for that record (see [Timestamp rendering](#timestamp-rendering)), not a WaveHouse rewrite — so a live event and a `/v1/query` read of the same row agree on spelling **by construction**, with no separate canonicalization step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). A column declared with a non-UTC zone is still rendered in UTC (the `Z` form), so the strings compare as the instants do. -**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause — a connection is never delivered a row the query path would hide for that role, and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable`, while other tenants' streams (and roles with no row filter) are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant's line is not served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` or `ALIAS` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert ClickHouse later rejects and parks on the dead-letter queue — a record the in-process parse accepted that the insert did not (a schema change between publish and insert, for instance), since an outage only delays a row and never drops it — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), but an expired token or changed claims take effect only when the client reconnects. +**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause on a server running that line's default settings — there, a connection is never delivered a row the query path would hide for that role (a server profile that changes a comparison setting can split the two; see [JWT claim templating](/access-control#jwt-claim-templating)), and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable`, while other tenants' streams (and roles with no row filter) are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant's line is not served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` or `ALIAS` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert ClickHouse later rejects and parks on the dead-letter queue — a record the in-process parse accepted that the insert did not (a schema change between publish and insert, for instance), since an outage only delays a row and never drops it — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), but an expired token or changed claims take effect only when the client reconnects. **CORS:** `/v1/stream` honors the request's tenant's `cors.allowed_origins` allowlist (settings directory) like every endpoint — the preflight included, which a browser sends without `X-Tenant-ID`, so over [a nested settings directory](/deployment#multi-tenant-deployments) the fronting proxy has to set the header on the `OPTIONS` too. Note that a **header-authenticated stream preflights before it connects** — `Authorization` is not CORS-safelisted — where a bare `EventSource` never preflighted at all: its request is not a `fetch()`, so Fetch's unsafe-request flag is never set and `Last-Event-ID` rides on the plain `GET`. Both headers are allow-listed, so an allowed origin connects *and* resumes cross-origin. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 10b6a992..1ccebe39 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -388,10 +388,12 @@ Client POST /v1/ops/query response shape stays "always an array." → Error: returns 4xx/5xx + plain-text error message + the X-ClickHouse-Exception-Code header. The handler classes it by - that code (chconn.Classify), not by the HTTP status ClickHouse - uses for nearly everything: bad SQL → 400, a missing grant → - 403, bad credentials → 502, an outage → 503 — with the trimmed - message, a `code` and `retryable` in the JSON error envelope. + that code (chconn.Classify), not by ClickHouse's HTTP status, + which does not say which kind of failure it was (on 26.8 a + syntax error is 400, an unknown table 404, a TIMEOUT_EXCEEDED + 408): bad SQL → 400, a missing grant → 403, bad credentials → + 502, an outage → 503 — with the trimmed message, a `code` and + `retryable` in the JSON error envelope. → Response carries Cache-Control: no-store so no downstream layer (browser, CDN, corp proxy) caches the result. ``` diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index b721abc7..ef82a3c2 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -269,7 +269,7 @@ Only the secrets are boot config, shared by every tenant. The verifier wiring | YAML Key | Env Var | Default | Description | | --- | --- | ------- | ----------- | | `auth.jwt_secret` | `WH_AUTH_JWT_SECRET` | *(empty)* | HMAC secret for JWT validation. Set this (or the settings directory's `auth.jwks_url`) so presented tokens are verified; see [Access Control](/access-control). Shared by every tenant; ignored for any tenant whose `jwks_url` is set. | -| `auth.operator_key` | `WH_AUTH_OPERATOR_KEY` | *(empty)* | Non-JWT operator credential. A request presenting it — in an `Authorization: Operator ` header, or the `X-Operator-Key` alias — is authorized as a full-access platform operator (the whole data plane *and* the `/v1/ops/*` management surface), independent of the JWT verifier, and it is honored even when no policy is adopted (break-glass). Empty disables it. Treat it as an admin secret. See below and [Access Control](/access-control). | +| `auth.operator_key` | `WH_AUTH_OPERATOR_KEY` | *(empty)* | Non-JWT operator credential. A request presenting it — in an `Authorization: Operator ` header, or the `X-Operator-Key` alias — is authorized as a full-access platform operator — the `/v1/ops/*` management surface always, and the entire data plane while a policy is adopted (with none, its data-plane requests are denied like anyone else's) — independent of the JWT verifier, so it still reaches `POST /v1/ops/settings/reload` when no policy is adopted (break-glass). Empty disables it. Treat it as an admin secret. See below and [Access Control](/access-control). | WaveHouse accepts only the signing algorithms matching the active verifier — `HS256`/`HS384`/`HS512` for the HMAC secret, or the asymmetric family (`RS*`/`ES*`/`PS*`/`EdDSA`) for JWKS — and validates the token's `alg` before any key is used, so your IdP must sign with one of these and `alg: none` is always rejected. diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index d57dad99..e07e2d68 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -188,9 +188,9 @@ WH_CH_PASSWORD= # resolves to the policy default_role). role_claim is a settings key too. WH_AUTH_JWT_SECRET= # Optional non-JWT operator credential (Authorization: Operator , or the -# X-Operator-Key alias): full-access -# admin for break-glass, honored even when no policy is adopted. Treat it -# as an admin secret — inject from your secret store, serve only over TLS. +# X-Operator-Key alias) for break-glass: /v1/ops/* always, and the whole data +# plane while a policy is adopted (with none, data-plane requests are denied +# like anyone else's). Treat it as an admin secret — inject from your secret store, serve only over TLS. WH_AUTH_OPERATOR_KEY= # Optional shared query cache for several instances (see Multiple instances diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index 8b541a7a..b8940209 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -173,7 +173,7 @@ These are the small targets behind `make dev` — useful directly when you want ### Running with observability -WaveHouse natively exports standard OpenTelemetry (OTLP) data — with no `OTEL_EXPORTER_OTLP_ENDPOINT` set, to `localhost:4317`, where these dashboards listen. Export is off by default (`otel.enabled: false` in `config.yaml`): the E2E fixture turns it on, so `make test-e2e` needs nothing, while for `make dev` set `otel.enabled: true` in `.config.local.yaml` or run `WH_OTEL_ENABLED=true make dev`. Rather than coupling a heavy observability database stack to the dev server, we provide three lightweight, single-container dashboard options. +WaveHouse natively exports standard OpenTelemetry (OTLP) data. These dashboards listen for plaintext OTLP on `localhost:4317`; the OpenTelemetry SDK's unset default dials that port over TLS, which they reject, so point it there explicitly with `OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317`. Export is off by default (`otel.enabled: false` in `config.yaml`): the E2E fixture turns it on and sets that endpoint, so `make test-e2e` needs nothing, while `make dev` needs both — `otel.enabled: true` in `.config.local.yaml` (or `WH_OTEL_ENABLED=true`) plus the endpoint variable, as in `WH_OTEL_ENABLED=true OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317 make dev`. Rather than coupling a heavy observability database stack to the dev server, we provide three lightweight, single-container dashboard options. You run these in a separate terminal tab alongside `make dev` or your test suites (`make test-e2e`). @@ -188,7 +188,7 @@ They block the terminal and stream logs; simply press `Ctrl+C` to instantly tear **Typical Workflow:** 1. Open Tab 1: run `make obs-aspire` (UI opens automatically) -2. Open Tab 2: run `WH_OTEL_ENABLED=true make dev` (or `make test-e2e`) +2. Open Tab 2: run `WH_OTEL_ENABLED=true OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317 make dev` (or `make test-e2e`) 3. View traces, metrics, and logs flowing into the UI instantly. No accounts or auth tokens required. ### Using the SDK against `make dev` From e7215e8607052e952d48e0a2f0e615a51b72d23b Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 10:56:26 -0400 Subject: [PATCH 52/70] perf(typelayer): compile role shapes outside the role cache lock roleTable held the per-table cache lock while compileRole ran (the role's CompileDDL plus the readBy probe compiles), against the package's own "compile outside every lock" rule. A shape key carries claim values, so a role that injects one value per end user (user_id = {{jwt.sub}}) misses on every request once a table has more users than roleCacheSize, and every miss compiled while holding the lock every role lookup on that table needs: the table's ingest serialized behind one compile at a time. The same happened after every rebind. A miss now registers a per-shape flight, compiles with the cache lock released, and re-locks to insert. Misses on different shapes compile in parallel; concurrent misses on one shape park on the flight and take the entry it lands. A hit still read-locks the projection under the cache lock, and the base table's read lock is held throughout, so eviction, rebind and Forget keep their ordering. A compile that panics withdraws its flight and releases the base table, so no waiter parks forever and a rebind still completes. BenchmarkRoleTable_ClaimChurn, 7-column table, 26.8 artifact, M4 Pro, 14 procs, A/B against the parent commit run back to back: claims=300 (past the cache, every lookup compiles): ~113 us/op -> ~19 us/op claims=64 (every lookup hits): ~0.57 us/op both Co-Authored-By: Claude Opus 5.5 --- internal/typelayer/roletable.go | 109 +++++++++-- internal/typelayer/roletable_test.go | 279 +++++++++++++++++++++++++++ 2 files changed, 373 insertions(+), 15 deletions(-) diff --git a/internal/typelayer/roletable.go b/internal/typelayer/roletable.go index f179d24b..d8800400 100644 --- a/internal/typelayer/roletable.go +++ b/internal/typelayer/roletable.go @@ -119,8 +119,10 @@ func (e *Engine) RoleTable(id tenant.ID, table string, shape RoleShape) (*Table, if shape.identity() { return base, nil // still read-locked; the caller's Release covers it } - rt, evicted, err := base.roleTable(shape) - base.Release() + rt, evicted, err := func() (*Table, *Table, error) { + defer base.Release() // a panicking compile must not leave a rebind waiting + return base.roleTable(shape) + }() // Closed after the cache lock and the base read lock are both gone: it // waits for the evicted shape's own readers, which are requests in flight. if evicted != nil { @@ -135,33 +137,54 @@ func (e *Engine) RoleTable(id tenant.ID, table string, shape RoleShape) (*Table, // roleTable is RoleTable's cache half, run under the base table's read lock. // It returns the projection read-locked, plus the entry its insertion evicted // (to be closed by the caller, outside the cache lock). +// +// A miss compiles outside the cache lock, which every role lookup on the +// table needs: a compile takes ~100 µs, and the shape key carries claim +// values, so a table whose roles inject one value per end user misses on +// every request once it has more users than roleCacheSize, and after every +// rebind. Misses on different shapes therefore compile in parallel, while +// concurrent misses on one shape wait for the first one's compile rather +// than repeat it. The base read lock stays held throughout, so a rebind +// waits for a compile in flight instead of closing the cache under it. func (t *Table) roleTable(shape RoleShape) (*Table, *Table, error) { key := shape.key(t.Generation) c := t.roles c.mu.Lock() - defer c.mu.Unlock() - - if el, hit := c.index[key]; hit { - c.order.MoveToFront(el) - e := el.Value.(*roleEntry) - if e.table == nil { - return nil, nil, &RoleRefused{Tenant: t.tenant, Table: t.Name, Cause: e.cause} + for { + if el, hit := c.index[key]; hit { + rt, err := t.hitLocked(el) + c.mu.Unlock() + return rt, nil, err } - // Taken while the base read lock is still held, so a rebind cannot be - // closing this projection underneath us. - e.table.mu.RLock() - return e.table, nil, nil + f := c.inflight[key] + if f == nil { + break + } + f.waiters++ + c.mu.Unlock() + <-f.done + // Landed, or withdrawn by a panic. Look again; a shape evicted in the + // meantime is compiled afresh below. + c.mu.Lock() } + f := &roleFlight{done: make(chan struct{})} + c.inflight[key] = f + c.mu.Unlock() - rt, cause := t.compileRole(shape) + rt, cause := c.compileFlight(t, shape, key, f) if cause != "" { slog.Error("chtypes could not compile a per-role schema; every insert for this role fails closed", "tenant", t.tenant, "table", t.Name, "generation", t.Generation, "allowed_columns", shape.Columns, "default_columns", slices.Sorted(maps.Keys(shape.Defaults)), "cause", cause) } + + c.mu.Lock() + defer c.mu.Unlock() el := c.order.PushFront(&roleEntry{key: key, table: rt, cause: cause}) c.index[key] = el + delete(c.inflight, key) + close(f.done) var evicted *Table if c.order.Len() > c.cap { evicted = c.evictOldestLocked() @@ -169,10 +192,45 @@ func (t *Table) roleTable(shape RoleShape) (*Table, *Table, error) { if rt == nil { return nil, evicted, &RoleRefused{Tenant: t.tenant, Table: t.Name, Cause: cause} } + // Under the cache lock, like a hit: nothing can have evicted it yet. rt.mu.RLock() return rt, evicted, nil } +// hitLocked answers a cached entry. The caller holds the cache lock and the +// base read lock. +func (t *Table) hitLocked(el *list.Element) (*Table, error) { + t.roles.order.MoveToFront(el) + e := el.Value.(*roleEntry) + if e.table == nil { + return nil, &RoleRefused{Tenant: t.tenant, Table: t.Name, Cause: e.cause} + } + // Taken under the cache lock, while the entry is still indexed, so no + // eviction is closing it; and under the base read lock, so no rebind is. + // Neither close has started, so this never waits. + e.table.mu.RLock() + return e.table, nil +} + +// compileFlight runs the compile for flight f, which this lookup registered. +// If the compile panics, f is withdrawn before the panic propagates, so its +// waiters wake, find neither an entry nor a flight, and compile for +// themselves rather than park forever. +func (c *roleCache) compileFlight(t *Table, shape RoleShape, key string, f *roleFlight) (*Table, string) { + landed := false + defer func() { + if !landed { + c.mu.Lock() + delete(c.inflight, key) + close(f.done) + c.mu.Unlock() + } + }() + rt, cause := c.compile(t, shape) + landed = true + return rt, cause +} + // compileRole builds and compiles the role's declaration list. It returns // (nil, cause) for every refusal. func (t *Table) compileRole(shape RoleShape) (*Table, string) { @@ -303,10 +361,31 @@ type roleCache struct { cap int order *list.List // front = most recently used index map[string]*list.Element + // inflight holds the shapes being compiled, outside mu, by a lookup that + // missed (see Table.roleTable). + inflight map[string]*roleFlight + // compile builds a shape's projection: Table.compileRole, which a test + // may wrap to hold one compile open. + compile func(*Table, RoleShape) (*Table, string) +} + +// roleFlight is one shape's compile in progress. done is closed, under the +// cache lock, once the shape's entry is in the cache or the compile panicked. +type roleFlight struct { + done chan struct{} + // waiters counts the lookups that parked on done; written under the + // cache lock, read by tests. + waiters int } func newRoleCache(capacity int) *roleCache { - return &roleCache{cap: capacity, order: list.New(), index: make(map[string]*list.Element)} + return &roleCache{ + cap: capacity, + order: list.New(), + index: make(map[string]*list.Element), + inflight: make(map[string]*roleFlight), + compile: (*Table).compileRole, + } } // capacity is the cache's bound, for a rebind's fresh cache; roleCacheSize diff --git a/internal/typelayer/roletable_test.go b/internal/typelayer/roletable_test.go index 72066470..28a6d128 100644 --- a/internal/typelayer/roletable_test.go +++ b/internal/typelayer/roletable_test.go @@ -3,8 +3,13 @@ package typelayer import ( "encoding/json" "errors" + "fmt" "runtime" + "strconv" + "sync" + "sync/atomic" "testing" + "time" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -427,3 +432,277 @@ func (c *roleCache) len() int { defer c.mu.Unlock() return c.order.Len() } + +// heldCompile wraps the orders table's role compile so the compile of one +// shape (Defaults tenant=claim) blocks until Release; entered is closed once +// that compile has started. compiles counts every compile. +type heldCompile struct { + claim string + entered chan struct{} + release chan struct{} + released sync.Once + compiles atomic.Int32 + // panics makes the held compile panic, once, when released. + panics atomic.Bool +} + +// Release lets the held compile finish. Also run at cleanup, before the +// engine closes: a failed assertion must not leave the engine's Close +// waiting on a compile nobody releases. +func (h *heldCompile) Release() { h.released.Do(func() { close(h.release) }) } + +func holdCompile(t *testing.T, eng *Engine, claim string) (*heldCompile, *roleCache) { + t.Helper() + h := &heldCompile{claim: claim, entered: make(chan struct{}), release: make(chan struct{})} + t.Cleanup(h.Release) + base, err := eng.Table(tenant.Default, "orders") + require.NoError(t, err) + defer base.Release() + c := base.roles + c.mu.Lock() + defer c.mu.Unlock() + var once sync.Once + c.compile = func(tbl *Table, shape RoleShape) (*Table, string) { + h.compiles.Add(1) + if shape.Defaults["tenant"] == h.claim { + once.Do(func() { close(h.entered) }) + <-h.release + if h.panics.CompareAndSwap(true, false) { + panic("compile panicked") + } + } + return tbl.compileRole(shape) + } + return h, c +} + +// within fails the test unless fn returns within the deadline. The deadline +// only bounds a failure; a pass never waits for it. +func within(t *testing.T, what string, fn func()) { + t.Helper() + done := make(chan struct{}) + go func() { + defer close(done) + fn() + }() + select { + case <-done: + case <-time.After(10 * time.Second): + t.Fatalf("%s did not return while another compile was held open", what) + } +} + +func claimShape(claim string) RoleShape { + return RoleShape{Defaults: map[string]string{"tenant": claim}} +} + +// injects asserts that a projection built from claimShape(claim) ingests a +// record omitting tenant with claim filled in. +func injects(t *testing.T, tbl *Table, claim string) { + t.Helper() + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"id":1,"amount":5}`+"\n")) + if assert.NoError(t, err) && assert.Len(t, batch.Rows, 1) { + assert.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) + assert.Equal(t, `[1, "`+claim+`", "", 5]`, string(batch.Rows[0].Line)) + } +} + +// TestRoleTable_MissesOnDifferentShapesDoNotSerialize: a miss compiles +// outside the cache lock, so while one shape's compile is held open, a hit +// and a miss on another shape of the same table both answer. With the +// compile under the lock, both would wait for the held one. +func TestRoleTable_MissesOnDifferentShapesDoNotSerialize(t *testing.T) { + eng := testEngine(t, ordersTable()) + warm, err := eng.RoleTable(tenant.Default, "orders", claimShape("warm")) + require.NoError(t, err) + warm.Release() + + h, _ := holdCompile(t, eng, "slow") + slow := make(chan struct{}) + go func() { + defer close(slow) + tbl, err := eng.RoleTable(tenant.Default, "orders", claimShape("slow")) + if assert.NoError(t, err) { + injects(t, tbl, "slow") + tbl.Release() + } + }() + <-h.entered + + for _, claim := range []string{"warm", "other"} { // a hit, then a miss on another shape + within(t, "a lookup of "+claim, func() { + tbl, err := eng.RoleTable(tenant.Default, "orders", claimShape(claim)) + if assert.NoError(t, err) { + injects(t, tbl, claim) + tbl.Release() + } + }) + } + + h.Release() + <-slow +} + +// TestRoleTable_ConcurrentMissesOnOneShapeCompileOnce: lookups that miss on a +// shape already compiling wait for that compile instead of starting their +// own, and all of them get the one projection it built. +func TestRoleTable_ConcurrentMissesOnOneShapeCompileOnce(t *testing.T) { + eng := testEngine(t, ordersTable()) + h, c := holdCompile(t, eng, "shared") + + const lookups = 8 + got := make([]*Table, lookups) + var wg sync.WaitGroup + for i := range got { + wg.Go(func() { + tbl, err := eng.RoleTable(tenant.Default, "orders", claimShape("shared")) + if assert.NoError(t, err) { + injects(t, tbl, "shared") + tbl.Release() + got[i] = tbl + } + }) + } + <-h.entered + key := claimShape("shared").key(1) + require.Eventually(t, func() bool { + c.mu.Lock() + defer c.mu.Unlock() + f := c.inflight[key] + return f != nil && f.waiters == lookups-1 + }, 10*time.Second, time.Millisecond, "every other lookup parks on the compile in flight") + + h.Release() + wg.Wait() + require.NotNil(t, got[0]) + for _, tbl := range got[1:] { + assert.Same(t, got[0], tbl) + } + assert.Equal(t, int32(1), h.compiles.Load()) + c.mu.Lock() + assert.Empty(t, c.inflight) + c.mu.Unlock() +} + +// TestRoleTable_PanickingCompileReleasesItsWaiters: a compile that panics +// withdraws its flight and releases the base table on the way out, so a +// lookup parked on it compiles for itself instead of waiting forever, and a +// rebind still completes. +func TestRoleTable_PanickingCompileReleasesItsWaiters(t *testing.T) { + eng := testEngine(t, ordersTable()) + h, c := holdCompile(t, eng, "boom") + h.panics.Store(true) + + panicked := make(chan any, 1) + go func() { + defer func() { panicked <- recover() }() + _, _ = eng.RoleTable(tenant.Default, "orders", claimShape("boom")) + }() + <-h.entered + waited := make(chan struct{}) + go func() { + defer close(waited) + tbl, err := eng.RoleTable(tenant.Default, "orders", claimShape("boom")) + if assert.NoError(t, err, "the waiter compiles the shape itself") { + injects(t, tbl, "boom") + tbl.Release() + } + }() + key := claimShape("boom").key(1) + require.Eventually(t, func() bool { + c.mu.Lock() + defer c.mu.Unlock() + f := c.inflight[key] + return f != nil && f.waiters == 1 + }, 10*time.Second, time.Millisecond) + + h.Release() + assert.Equal(t, "compile panicked", <-panicked) + <-waited + assert.Equal(t, int32(2), h.compiles.Load()) + + changed := ordersTable() + changed.Columns[3].Type = "UInt32" + within(t, "a rebind after the panic", func() { + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{changed}) + }) +} + +// TestRoleTable_ConcurrentLookupsEvictionAndRebind races lookups on shared +// and distinct shapes, through a cache small enough to evict on most +// misses, against rebinds. Under -race it pins the flight bookkeeping, the +// evicted handle closing outside the cache lock, and a rebind's closeAll +// waiting out compiles in flight; every lookup answers a projection that +// injects its own claim. +func TestRoleTable_ConcurrentLookupsEvictionAndRebind(t *testing.T) { + eng := testEngine(t, ordersTable()) + base, err := eng.Table(tenant.Default, "orders") + require.NoError(t, err) + base.Release() + base.mu.Lock() + base.roles = newRoleCache(4) // a rebind keeps the capacity + base.mu.Unlock() + + var wg sync.WaitGroup + for g := range 12 { + wg.Go(func() { + for i := range 25 { + claim := "c" + strconv.Itoa((g*7+i)%10) + tbl, err := eng.RoleTable(tenant.Default, "orders", claimShape(claim)) + if !assert.NoError(t, err) { + return + } + injects(t, tbl, claim) + tbl.Release() + } + }) + } + wg.Go(func() { + for i := range 6 { + changed := ordersTable() + if i%2 == 0 { + changed.Columns[3].Type = "UInt32" + } + eng.Bind(tenant.Default, testServerVersion, "UTC", []*discovery.TableSchema{changed}) + } + }) + wg.Wait() + + base, err = eng.Table(tenant.Default, "orders") + require.NoError(t, err) + defer base.Release() + assert.Equal(t, uint64(7), base.Generation) + assert.LessOrEqual(t, base.roles.len(), 4) + base.roles.mu.Lock() + assert.Empty(t, base.roles.inflight) + base.roles.mu.Unlock() +} + +// BenchmarkRoleTable_ClaimChurn is a per-role lookup whose shape carries an +// end user's claim (an insert check such as user_id = {{jwt.sub}}, injected +// as a column default), so each distinct claim is a distinct shape. +// claims=64 sits inside roleCacheSize and measures the hit; claims=300 +// cycles past it, so every lookup compiles. It runs in parallel because a +// miss compiles outside the cache lock: misses on different shapes should +// scale with cores rather than queue behind one another. +func BenchmarkRoleTable_ClaimChurn(b *testing.B) { + eng := testEngine(b, rowsTable()) + for _, n := range []int{64, 300} { + shapes := make([]RoleShape, n) + for i := range shapes { + shapes[i] = RoleShape{Defaults: map[string]string{"tenant": "user-" + strconv.Itoa(i)}} + } + b.Run(fmt.Sprintf("claims=%d", n), func(b *testing.B) { + var next atomic.Int64 + b.RunParallel(func(pb *testing.PB) { + for pb.Next() { + tbl, err := eng.RoleTable(tenant.Default, "rows", shapes[int(next.Add(1))%n]) + if err != nil { + b.Fatal(err) + } + tbl.Release() + } + }) + }) + } +} From f36751130f438e86061ef032e6386e42067cb6c0 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 10:59:12 -0400 Subject: [PATCH 53/70] perf(typelayer): give every handle the whole filter cache budget The 4096-filter budget was split across the pool's limit, so on a host with 8 or more cores each handle held 512 filters. A filter answers only on the handle it was compiled against, so the slot an evaluation lands on has to hold the whole working set: past 512 distinct (expression, claims) pairs per table every subscriber evaluation recompiled, on the tenant's serial hub-bridge callback, and a bigger host hit it sooner. Every slot now has the full filterCacheSize. The bound per table is the pool's limit times that; measured on the 26.8 artifact (4096 compiled on one handle, darwin arm64) a one-clause filter is ~10 KiB resident and a two-clause one with a three-value IN ~27 KiB, so a slot of one-clause filters is ~43 MiB and a full pool of 8 ~340 MiB, reached only with 4096 distinct pairs live on one table and evaluations on all 8 handles. To keep the copies proportional to contention rather than to the pool's size, acquire takes the lowest idle slot instead of rotating: a serial evaluation stream stays on one slot and compiles each filter once, where rotating compiled it on every slot. Only the all-busy fallback rotates. The eviction race stays fail-closed: a filter evicted and closed between lookup and Eval answers "filter is closed" (the SDK checks under the schema lock Close also takes), which withholds the row as a decline. BenchmarkVisible_ClaimChurn, 7-column table, pool grown to 8, M4 Pro, A/B against the parent commit run back to back: subscribers=600: ~22.9 us/eval -> ~3.4 us/eval subscribers=64: ~3.6 us/eval both Co-Authored-By: Claude Opus 5.5 --- internal/typelayer/filter.go | 27 +++++++-- internal/typelayer/filter_test.go | 94 ++++++++++++++++++++++++++---- internal/typelayer/pool.go | 41 ++++++------- internal/typelayer/tenancy_test.go | 38 +++++++++++- 4 files changed, 164 insertions(+), 36 deletions(-) diff --git a/internal/typelayer/filter.go b/internal/typelayer/filter.go index 5de3703a..16cf69f6 100644 --- a/internal/typelayer/filter.go +++ b/internal/typelayer/filter.go @@ -15,11 +15,23 @@ import ( "github.com/Wave-RF/WaveHouse/internal/policy" ) -// filterCacheSize bounds the compiled filters held per table. A filter handle -// is identified by (expression, bound values), and the values come from tenant -// claims, so an unbounded cache is a memory/CPU denial of service. The budget -// is split across the handle pool's limit (see newPool), so this is the -// table's total, not each slot's. +// filterCacheSize bounds the compiled filters each handle of a table holds. A +// filter handle is identified by (expression, bound values), and the values +// come from tenant claims, so an unbounded cache is a memory/CPU denial of +// service. +// +// The bound is per slot, not a table budget split across the pool: a filter +// answers only on the handle it was compiled against, so the slot an +// evaluation lands on must hold the whole working set, and a split would +// shrink it as the host's cores grow (512 a slot in a pool of 8). Per table +// that makes the bound the pool's limit times this. A one-clause filter costs +// ~10 KiB resident and a two-clause one with a three-value IN ~27 KiB (each +// measured as 4096 compiled on one handle of the 26.8 artifact, darwin arm64): +// ~43 MiB for a slot of one-clause filters, ~340 MiB for a full pool of 8, +// reached only with 4096 distinct pairs live on the table and evaluations +// landing on all 8 handles. A handle past the first exists only under +// contention, and pool.acquire keeps a serial stream of evaluations on one +// slot, so a quiet table stays at one slot's figure. const filterCacheSize = 4096 // Predicate is one resolved row-filter clause. Values are the canonical strings @@ -153,6 +165,11 @@ func (r *Row) VisibleWithReason(preds []Predicate) (bool, string) { if filter == nil { return false, ReasonDecline } + // Another call on this slot can evict and close the filter between the + // lookup and this Eval. The SDK's Eval and Close take the same schema + // lock and Eval checks for a closed filter under it, so Eval answers + // "filter is closed" instead of touching freed memory, and that error + // withholds the row like any decline: fail closed. res, err := filter.Eval(r.block) if err != nil || res.Outcome != chtypes.FilterOK || len(res.Verdicts) == 0 { return false, ReasonDecline diff --git a/internal/typelayer/filter_test.go b/internal/typelayer/filter_test.go index a0f23331..ae14e19b 100644 --- a/internal/typelayer/filter_test.go +++ b/internal/typelayer/filter_test.go @@ -4,6 +4,7 @@ import ( "encoding/json" "fmt" "math/big" + "runtime" "strconv" "strings" "sync" @@ -12,6 +13,8 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + "github.com/wave-rf/chtypes/go/chtypes" + "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/tenant" ) @@ -572,23 +575,47 @@ func TestFilterCache_BoundedUnderTenantValueChurn(t *testing.T) { assert.True(t, row.Visible([]Predicate{{Column: "tenant", Op: "=", Values: []string{"acme"}}})) } -// TestFilterCache_BudgetIsSplitAcrossTheHandlePool: the 4096 bound is what -// makes the cache not a DoS, and it must be a bound on the TABLE — a pool of -// handles must not multiply it. -func TestFilterCache_BudgetIsSplitAcrossTheHandlePool(t *testing.T) { +// TestFilterCache_EverySlotHoldsTheWholeBudget: a filter answers only on its +// own handle, so the slot an evaluation lands on must hold the whole working +// set. Every slot of a grown pool is bounded by filterCacheSize itself, not +// by a share of it, and one slot holds more distinct filters than an +// eight-way split gave it (512) without recompiling any of them. +func TestFilterCache_EverySlotHoldsTheWholeBudget(t *testing.T) { + defer runtime.GOMAXPROCS(runtime.GOMAXPROCS(maxPoolSize)) eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") require.NoError(t, err) defer tbl.Release() - // The pool starts at one handle and may grow to its limit; the budget - // holds at the limit, not just at today's size. p := tbl.pool assert.Len(t, p.list(), 1, "one handle after Bind; the rest are compiled on contention") - assert.Equal(t, int64(poolSize()), p.limit.Load()) - assert.LessOrEqual(t, p.perSlot*poolSize(), filterCacheSize) + held := make([]*schemaSlot, 0, maxPoolSize) + for range maxPoolSize { + held = append(held, p.acquire()) + } + for _, s := range held { + p.release(s) + } + require.Len(t, p.list(), maxPoolSize) for _, s := range p.list() { - assert.Equal(t, p.perSlot, s.filters.cap) + assert.Equal(t, filterCacheSize, s.filters.cap) + } + + s := p.first() + filter := func(i int) *chtypes.LoadedFilter { + expr, params, ok := tbl.render([]Predicate{{Column: "tenant", Op: "=", Values: []string{"user-" + strconv.Itoa(i)}}}) + require.True(t, ok) + return tbl.filterOn(s, expr, params) + } + n := filterCacheSize/maxPoolSize + 1 + compiled := make([]*chtypes.LoadedFilter, n) + for i := range compiled { + compiled[i] = filter(i) + require.NotNil(t, compiled[i]) + } + assert.Equal(t, n, s.filters.len()) + for i, f := range compiled { + assert.Same(t, f, filter(i), "filter %d is still cached, not recompiled", i) } } @@ -641,7 +668,7 @@ func TestVisible_ConcurrentSubscribers(t *testing.T) { // TestTable_ConcurrentAcrossThePool exercises every entry point on one table // from many goroutines at once. Under -race it is the pin that the pool's -// round-robin, the per-slot filter caches and the shared block parsing are +// slot choice, the per-slot filter caches and the shared block parsing are // safe; a slot chosen per call rather than per Row would show up here as a // filter and a block on different handles. func TestTable_ConcurrentAcrossThePool(t *testing.T) { @@ -684,3 +711,50 @@ func TestTable_ConcurrentAcrossThePool(t *testing.T) { } wg.Wait() } + +// BenchmarkVisible_ClaimChurn is the stream's fan-out: one parsed event, then +// one evaluation per subscriber, each subscriber's claim a distinct filter, +// on a pool grown to its limit. The event's handle must hold the whole set: +// subscribers=600 is past the 512 a slot got when the table's budget was +// split across a pool of 8, and inside filterCacheSize. +func BenchmarkVisible_ClaimChurn(b *testing.B) { + eng := testEngine(b, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(b, err) + defer tbl.Release() + + // Grow the pool to its limit, as a busy table's would be. + p := tbl.pool + held := make([]*schemaSlot, 0, poolSize()) + for range poolSize() { + held = append(held, p.acquire()) + } + for _, s := range held { + p.release(s) + } + + for _, n := range []int{64, 600} { + preds := make([][]Predicate, n) + for i := range preds { + preds[i] = []Predicate{{Column: "tenant", Op: "=", Values: []string{"user-" + strconv.Itoa(i)}}} + } + event := func(b *testing.B) { + row, err := tbl.ParseRow(tbl.WireColumns, []byte(sampleRow)) + if err != nil { + b.Fatal(err) + } + for _, pred := range preds { + row.Visible(pred) + } + row.Close() + } + b.Run(fmt.Sprintf("subscribers=%d", n), func(b *testing.B) { + event(b) // compiles the set on the slot a serial event lands on + b.ResetTimer() + for range b.N { + event(b) + } + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*n), "ns/eval") + }) + } +} diff --git a/internal/typelayer/pool.go b/internal/typelayer/pool.go index 16942c05..5fe26ec7 100644 --- a/internal/typelayer/pool.go +++ b/internal/typelayer/pool.go @@ -108,9 +108,8 @@ func closeSlots(slots []*schemaSlot) { // only closed under that Table's write lock or once nothing can reach it, so // growth and close never overlap. type pool struct { - lib *chtypes.Library - ddl string - perSlot int // filter-cache capacity of each slot + lib *chtypes.Library + ddl string // limit is the most slots the pool will hold; lowered to the current size // if a growth compile is ever refused, so a refusal costs one compile. limit atomic.Int64 @@ -124,14 +123,9 @@ type pool struct { // newPool compiles a pool's first handle. A refusal is the whole shape's: the // handles are interchangeable by construction, so there is no half-pool. -// -// The per-table filter budget is SPLIT across the pool's limit rather than -// multiplied by it: values are tenant-controlled and baked into a compiled -// handle, so the bound that makes the cache not-a-DoS has to be a bound on -// the table. func newPool(lib *chtypes.Library, ddl string, limit int) (*pool, string) { limit = max(limit, 1) - p := &pool{lib: lib, ddl: ddl, perSlot: max(filterCacheSize/limit, 1)} + p := &pool{lib: lib, ddl: ddl} p.limit.Store(int64(limit)) first, err := p.compile() if err != nil { @@ -146,7 +140,7 @@ func (p *pool) compile() (*schemaSlot, error) { if err != nil { return nil, err } - return &schemaSlot{schema: schema, filters: newFilterCache(p.perSlot)}, nil + return &schemaSlot{schema: schema, filters: newFilterCache(filterCacheSize)}, nil } // list is the current slots, oldest first. @@ -156,17 +150,22 @@ func (p *pool) list() []*schemaSlot { return *p.slots.Load() } func (p *pool) first() *schemaSlot { return p.list()[0] } // acquire picks the slot one call (or one parsed Row) runs on and marks it -// busy until release. An idle slot is taken first; when every slot is busy and -// the pool is below its limit, the caller compiles one more and takes it; a -// caller that cannot grow the pool shares a busy slot, whose handle serializes -// the two calls. Growing on contention rather than at Bind is what keeps a -// quiet table, of a quiet tenant, at one handle. +// busy until release. The lowest idle slot is taken first; when every slot is +// busy and the pool is below its limit, the caller compiles one more and +// takes it; a caller that cannot grow the pool shares a busy slot, whose +// handle serializes the two calls. Growing on contention rather than at Bind +// is what keeps a quiet table, of a quiet tenant, at one handle. +// +// The LOWEST idle slot rather than a rotating one, because a filter answers +// only on the slot it was compiled on: a serial run of calls — a tenant's +// stream evaluating every subscriber's filter, event after event — stays on +// one slot and compiles each filter once, where rotating would compile it on +// every slot and keep a copy in each slot's cache. Copies then grow with real +// contention rather than with the pool's size. func (p *pool) acquire() *schemaSlot { slots := p.list() - n := uint64(len(slots)) - start := p.next.Add(1) % n - for i := range n { - if s := slots[(start+i)%n]; s.busy.CompareAndSwap(0, 1) { + for _, s := range slots { + if s.busy.CompareAndSwap(0, 1) { return s } } @@ -177,7 +176,9 @@ func (p *pool) acquire() *schemaSlot { return s } } - s := slots[start] + // Every slot busy and the pool full: share one, rotating so the sharing + // spreads. + s := slots[p.next.Add(1)%uint64(len(slots))] s.busy.Add(1) return s } diff --git a/internal/typelayer/tenancy_test.go b/internal/typelayer/tenancy_test.go index 3212abe3..1e52fd0a 100644 --- a/internal/typelayer/tenancy_test.go +++ b/internal/typelayer/tenancy_test.go @@ -247,10 +247,46 @@ func TestPool_GrowsLazilyToItsLimit(t *testing.T) { assert.Len(t, p.list(), 3) p.release(e) for _, s := range p.list() { - assert.Equal(t, p.perSlot, s.filters.cap, "every grown slot gets the split filter budget") + assert.Equal(t, filterCacheSize, s.filters.cap, "every grown slot gets the whole filter budget") } } +// TestPool_PrefersTheLowestIdleSlot: a filter answers only on the slot it +// was compiled on, so a serial run of calls must keep landing on one slot +// (and compile each filter once) rather than rotate through the pool and +// compile it on every slot. A busy slot is skipped, not waited for. +func TestPool_PrefersTheLowestIdleSlot(t *testing.T) { + eng := testEngine(t, rowsTable()) + tbl, err := eng.Table(tenant.Default, "rows") + require.NoError(t, err) + defer tbl.Release() + + p, cause := newPool(tbl.lib, tbl.pool.ddl, 3) + require.Empty(t, cause) + t.Cleanup(p.close) + grown := []*schemaSlot{p.acquire(), p.acquire(), p.acquire()} + for _, s := range grown { + p.release(s) + } + slots := p.list() + require.Len(t, slots, 3) + + for range 10 { + s := p.acquire() + assert.Same(t, slots[0], s, "a serial caller stays on the first slot") + p.release(s) + } + + busy := p.acquire() + require.Same(t, slots[0], busy) + for range 5 { + s := p.acquire() + assert.Same(t, slots[1], s, "the lowest idle slot, past the busy one") + p.release(s) + } + p.release(busy) +} + // TestTable_PoolGrowsUnderConcurrentHolders drives growth through the public // path: a parsed Row keeps its handle busy until Close. A burst of concurrent // parses grows the pool (a caller that loses the race to grow shares a busy From 6cb72bd779f7ab432686a5bec8532d3d39ff7c23 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 10:59:19 -0400 Subject: [PATCH 54/70] chore: reflow comments left over-long by the last round The record-count note in the ingest handler, the chtypes fetch step in setup-env (which also said "against chtypes.lock" twice), a policy test comment and the ClickHouse status comment in the query handler, back to the ~80-column comment style the rest of each file uses. No change in what they say. Co-Authored-By: Claude Opus 5.5 --- .github/actions/setup-env/action.yml | 6 +++--- internal/api/ingest.go | 11 ++++++----- internal/api/query.go | 3 +-- internal/policy/policy_test.go | 5 +++-- 4 files changed, 13 insertions(+), 12 deletions(-) diff --git a/.github/actions/setup-env/action.yml b/.github/actions/setup-env/action.yml index 0e3ce65a..37542f6e 100644 --- a/.github/actions/setup-env/action.yml +++ b/.github/actions/setup-env/action.yml @@ -211,9 +211,9 @@ runs: restore-keys: | chtypes-abi6-${{ runner.os }}-${{ runner.arch }}- - # Runs on both cache hit and miss: scripts/fetch-chtypes.sh (always a - # --frozen fetch against chtypes.lock) is cheap on a hit (a manifest check against chtypes.lock, not a - # re-download — see the script) and it is what turns a restored-but- + # Runs on both cache hit and miss: scripts/fetch-chtypes.sh always fetches + # frozen against chtypes.lock, is cheap on a hit (a manifest check, not a + # re-download — see the script), and is what turns a restored-but- # unverified cache entry back into a hash-checked one every run. - name: Fetch pinned chtypes artifact if: ${{ inputs.chtypes == 'true' }} diff --git a/internal/api/ingest.go b/internal/api/ingest.go index b5943c41..33c4f1c4 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -375,11 +375,12 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { } records = n } - // Otherwise a single-object body is one record (concatenated objects after - // it are neither answered nor published, as they always have been — declare - // NDJSON to batch them, #561; chtypes still parses them, so one cut off - // mid-record can turn the answer into a decline), and a line-framed body has at least the record its first byte - // starts. The real count is chtypes' own, taken once it has answered. + // Otherwise a single-object body is one record, and a line-framed body has + // at least the record its first byte starts. Concatenated objects after a + // single object are neither answered nor published, as they always have + // been (declare NDJSON to batch them, #561), but chtypes still parses + // them, so one cut off mid-record can turn the answer into a decline. The + // real count is chtypes' own, taken once it has answered. guard := h.policyCheckGuard(ctx, table, role, schema, perms) shape, preds, checkColumns, abort := h.insertShape(ctx, table, role, schema, perms, guard) diff --git a/internal/api/query.go b/internal/api/query.go index b622e4f9..e1b675a1 100644 --- a/internal/api/query.go +++ b/internal/api/query.go @@ -296,8 +296,7 @@ func (h *QueryHandler) Handle(w http.ResponseWriter, r *http.Request) { // The status ClickHouse sends does not say which kind of failure it // was (on 26.8 a syntax error is 400, an unknown table 404, a // TIMEOUT_EXCEEDED 408); the exception code it sends with it does - // (#403). The message is - // ClickHouse's own text, verbatim. + // (#403). The message is ClickHouse's own text, verbatim. chErr := chconn.NewHTTPError(&http.Response{StatusCode: resp.StatusCode, Header: resp.Header, Body: io.NopCloser(bytes.NewReader(body))}) msg := strings.TrimSpace(string(body)) if msg == "" { diff --git a/internal/policy/policy_test.go b/internal/policy/policy_test.go index 39615394..4eb1233f 100644 --- a/internal/policy/policy_test.go +++ b/internal/policy/policy_test.go @@ -442,8 +442,9 @@ func TestResolveTemplate(t *testing.T) { {"large integer claim binds exactly", "{{ jwt.big }}", "12345678901234567890", true}, // Numeric claims bind in canonical decimal form, not the token's // spelling — bound verbatim, "1.0"/"1e3" would match nothing on an - // integer column, which reads only the canonical spelling. A magnitude only JSON can hold fails - // closed like any other unresolvable claim. + // integer column, which reads only the canonical spelling. A + // magnitude only JSON can hold fails closed like any other + // unresolvable claim. {"float spelling binds canonically", "{{ jwt.price }}", "1", true}, {"exponent spelling binds canonically", "{{ jwt.exp3 }}", "1000", true}, {"beyond-float64 number fails closed", "{{ jwt.huge }}", "", false}, From e8ad99a2fd5c1aa1a93b7c2d4933ea6530fd36a8 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 11:16:35 -0400 Subject: [PATCH 55/70] docs: name roweval.go in the stream package and who sets the e2e OTLP endpoint Co-Authored-By: Claude Opus 5.5 --- docs/src/content/docs/architecture.md | 1 + docs/src/content/docs/development.md | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 1ccebe39..77611239 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -107,6 +107,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)) lives next to the keepalive primitives it shares. One abstraction per file. - **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row. Visibility itself is decided by `internal/typelayer` (`Table.ParseRow` once per event, on the event's tenant's table set, then `Row.Visible` per subscriber) — the same ClickHouse parsing and comparison semantics the server's own `WHERE` clause applies, for every column type, rather than a hand-written per-type comparator; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (one the inserting role cannot write, or a `MATERIALIZED`/`ALIAS` column; reason `decline`, though `/v1/query` still returns the row), a schema drift between the event and the live table, or an engine that is unavailable for that tenant all withhold the row rather than guessing. Each row withheld this way increments `wavehouse_sse_rows_withheld_total{table,role,reason}`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and, under a row filter, preparing one parsed row per replayed event (closed at once, since a gap-fill can run for thousands of events). `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. +- **roweval.go** — `RowEvaluator` / `RowView`, the one place stream row visibility under a role's row `filter` is decided: `Prepare` parses an event once through `internal/typelayer`, and the view answers per subscriber. A nil evaluator fails closed, and `WithheldReason` maps a prepare error to the `reason` label of `wavehouse_sse_rows_withheld_total` (`filter`, `error`, `decline`, `unavailable`, `drift`). - **subscriber.go** — `Subscriber`, the per-connection handle. It carries the connection's JWT claims, fixed at construction (`NewSubscriber(claims, metrics)`, no setter) — the claims the `Hub` resolves a role's row-level `filter` against, and immutability is what makes the fan-out's unsynchronized claims read race-free structurally. It owns a single ready-to-write outbound queue of `Frame`s (each tagged with its `kind`, so the handler labels the write where it happens): producers — the keepalive wheel and the event `Hub` — fan frames in with `Send` (non-blocking; a full queue drops, and `Send` itself counts the drop by frame kind, so no producer can forget to), and the handler drains `Frames()` to the client verbatim. The queue is sized for buffering live events (cap 64, up from the keepalive-only cap 1; #152 will make it a knob), and `Evict` closes its `Evicted()` channel, once, for the handler to end the stream: the `Hub`'s `Prune` does for a tenant no longer served, and the slow-consumer follow-up will for a wedged consumer. - **bucket.go** — `Bucket`, the reusable fan-out primitive: a concurrency-safe set of subscribers. `Push` fans one `Frame` to every member fire-and-forget — the keepalive wheel's ring is its only caller now that both `Hub` paths iterate `Snapshot`, since the schema announcement is per connection even where the projection is shared per role; `Snapshot` exposes the members so the event `Hub` can evaluate row visibility per subscriber before sending (drop counting lives in `Send` itself). The `Hub` holds one `Bucket` per `(topic, role)` so a projected frame is built once and sent to every member instead of re-projected per subscriber. - **heartbeat.go** — The keepalive wheel (`Heartbeater`). A single process-wide ticker fans a minimal `:` comment across the ring of `Bucket`s, waking ~1/N of live streams per tick so the writes don't synchronize. The effective per-connection keepalive period is `stream.keepalive_interval` in the settings directory (the wheel ticks every `keepalive_interval ÷ keepalive_buckets`, so one rotation spans the interval; a reload calls `Reconfigure`, which rebuilds the ring in place with every live subscriber carried over); the owning handler goroutine does the actual write, so the shared ticker never touches a `ResponseWriter` directly. diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index b8940209..0187607d 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -173,7 +173,7 @@ These are the small targets behind `make dev` — useful directly when you want ### Running with observability -WaveHouse natively exports standard OpenTelemetry (OTLP) data. These dashboards listen for plaintext OTLP on `localhost:4317`; the OpenTelemetry SDK's unset default dials that port over TLS, which they reject, so point it there explicitly with `OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317`. Export is off by default (`otel.enabled: false` in `config.yaml`): the E2E fixture turns it on and sets that endpoint, so `make test-e2e` needs nothing, while `make dev` needs both — `otel.enabled: true` in `.config.local.yaml` (or `WH_OTEL_ENABLED=true`) plus the endpoint variable, as in `WH_OTEL_ENABLED=true OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317 make dev`. Rather than coupling a heavy observability database stack to the dev server, we provide three lightweight, single-container dashboard options. +WaveHouse natively exports standard OpenTelemetry (OTLP) data. These dashboards listen for plaintext OTLP on `localhost:4317`; the OpenTelemetry SDK's unset default dials that port over TLS, which they reject, so point it there explicitly with `OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317`. Export is off by default (`otel.enabled: false` in `config.yaml`): the E2E fixture turns it on and the orchestrator behind `make test-e2e` sets that endpoint, so `make test-e2e` needs nothing (a server you start yourself with the fixture config, below, needs `OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317` like `make dev`), while `make dev` needs both — `otel.enabled: true` in `.config.local.yaml` (or `WH_OTEL_ENABLED=true`) plus the endpoint variable, as in `WH_OTEL_ENABLED=true OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317 make dev`. Rather than coupling a heavy observability database stack to the dev server, we provide three lightweight, single-container dashboard options. You run these in a separate terminal tab alongside `make dev` or your test suites (`make test-e2e`). From f09d96ff23d78ef60f43bccd0dda42f09d02b3f7 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 11:26:11 -0400 Subject: [PATCH 56/70] docs: show ClickHouse's escaped slashes in raw stream and pipe output Co-Authored-By: Claude Opus 5.5 --- docs/src/content/docs/api.md | 6 ++++-- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/getting-started.md | 2 +- docs/src/content/docs/pipes.mdx | 2 +- 4 files changed, 7 insertions(+), 5 deletions(-) diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 1f25f848..efa35bd8 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -674,12 +674,14 @@ event: schema data: {"table_name":"clicks","columns":["page","button","score","received_timestamp"]} id: 2026-03-24T12:00:00.123Z -data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:00.123Z","row":["/home","signup",42.5,"2026-03-24T11:59:58.512Z"]} +data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:00.123Z","row":["\/home","signup",42.5,"2026-03-24T11:59:58.512Z"]} id: 2026-03-24T12:00:01.456Z -data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,"2026-03-24T12:00:01.456Z"]} +data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["\/pricing","cta",7,"2026-03-24T12:00:01.456Z"]} ``` +String values are ClickHouse's JSON rendering, which escapes `/` as `\/` — valid JSON that any parser reads back as `/`. + A raw consumer must keep the most recent announced column list and zip each `row` against it; a column the record omitted still has its slot, holding its evaluated `DEFAULT` (or the type's default — `null` only on a `Nullable` column with none), so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you: `.stream()` and `.liveQuery()` zip each row into an object. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. Each SSE connection is bound to a single `?table=`; to consume multiple tables, open one connection per table. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 77611239..d10b0e24 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -107,7 +107,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)) lives next to the keepalive primitives it shares. One abstraction per file. - **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row. Visibility itself is decided by `internal/typelayer` (`Table.ParseRow` once per event, on the event's tenant's table set, then `Row.Visible` per subscriber) — the same ClickHouse parsing and comparison semantics the server's own `WHERE` clause applies, for every column type, rather than a hand-written per-type comparator; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (one the inserting role cannot write, or a `MATERIALIZED`/`ALIAS` column; reason `decline`, though `/v1/query` still returns the row), a schema drift between the event and the live table, or an engine that is unavailable for that tenant all withhold the row rather than guessing. Each row withheld this way increments `wavehouse_sse_rows_withheld_total{table,role,reason}`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and, under a row filter, preparing one parsed row per replayed event (closed at once, since a gap-fill can run for thousands of events). `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. -- **roweval.go** — `RowEvaluator` / `RowView`, the one place stream row visibility under a role's row `filter` is decided: `Prepare` parses an event once through `internal/typelayer`, and the view answers per subscriber. A nil evaluator fails closed, and `WithheldReason` maps a prepare error to the `reason` label of `wavehouse_sse_rows_withheld_total` (`filter`, `error`, `decline`, `unavailable`, `drift`). +- **roweval.go** — `RowEvaluator` / `RowView`, the one place stream row visibility under a role's row `filter` is decided: `Prepare` parses an event once through `internal/typelayer`, and the view answers per subscriber. A nil evaluator fails closed, and `WithheldReason` maps a prepare error to its `reason` label of `wavehouse_sse_rows_withheld_total` (`unavailable`, `drift` or `error`); the view's `Visible` supplies `filter` and `decline`. - **subscriber.go** — `Subscriber`, the per-connection handle. It carries the connection's JWT claims, fixed at construction (`NewSubscriber(claims, metrics)`, no setter) — the claims the `Hub` resolves a role's row-level `filter` against, and immutability is what makes the fan-out's unsynchronized claims read race-free structurally. It owns a single ready-to-write outbound queue of `Frame`s (each tagged with its `kind`, so the handler labels the write where it happens): producers — the keepalive wheel and the event `Hub` — fan frames in with `Send` (non-blocking; a full queue drops, and `Send` itself counts the drop by frame kind, so no producer can forget to), and the handler drains `Frames()` to the client verbatim. The queue is sized for buffering live events (cap 64, up from the keepalive-only cap 1; #152 will make it a knob), and `Evict` closes its `Evicted()` channel, once, for the handler to end the stream: the `Hub`'s `Prune` does for a tenant no longer served, and the slow-consumer follow-up will for a wedged consumer. - **bucket.go** — `Bucket`, the reusable fan-out primitive: a concurrency-safe set of subscribers. `Push` fans one `Frame` to every member fire-and-forget — the keepalive wheel's ring is its only caller now that both `Hub` paths iterate `Snapshot`, since the schema announcement is per connection even where the projection is shared per role; `Snapshot` exposes the members so the event `Hub` can evaluate row visibility per subscriber before sending (drop counting lives in `Send` itself). The `Hub` holds one `Bucket` per `(topic, role)` so a projected frame is built once and sent to every member instead of re-projected per subscriber. - **heartbeat.go** — The keepalive wheel (`Heartbeater`). A single process-wide ticker fans a minimal `:` comment across the ring of `Bucket`s, waking ~1/N of live streams per tick so the writes don't synchronize. The effective per-connection keepalive period is `stream.keepalive_interval` in the settings directory (the wheel ticks every `keepalive_interval ÷ keepalive_buckets`, so one rotation spans the interval; a reload calls `Reconfigure`, which rebuilds the ring in place with every live subscriber carried over); the owning handler goroutine does the actual write, so the shared ticker never touches a `ResponseWriter` directly. diff --git a/docs/src/content/docs/getting-started.md b/docs/src/content/docs/getting-started.md index 1e231737..16425029 100644 --- a/docs/src/content/docs/getting-started.md +++ b/docs/src/content/docs/getting-started.md @@ -95,7 +95,7 @@ curl -N "http://localhost:8080/v1/stream?table=clicks" curl -N "http://localhost:8080/v1/stream?table=clicks&since=2026-03-24T11:00:00Z" ``` -Rows arrive **positionally**, so raw `curl` output looks like `"row":["/home","signup",42.5,"2026-03-24T11:59:58.512Z"]` rather than named fields. The stream sends an `event: schema` frame before the first row and again when the column list changes, and a raw client must pair each row against the **most recent** frame rather than the first. +Rows arrive **positionally**, so raw `curl` output looks like `"row":["\/home","signup",42.5,"2026-03-24T11:59:58.512Z"]` rather than named fields (strings are ClickHouse's JSON rendering, which escapes `/` as `\/`; any JSON parser reads it back as `/`). The stream sends an `event: schema` frame before the first row and again when the column list changes, and a raw client must pair each row against the **most recent** frame rather than the first. That re-announcement is not guaranteed in one case: after a gap-fill across a column change, live rows can arrive without a fresh frame ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). Drop a row whose length disagrees with the last announced list rather than zipping it, and reconnect to resynchronize — an arity check cannot see a same-length change such as a `RENAME COLUMN`. See [the wire format](/api#get-v1stream--server-sent-events-stream) for the frame sequence and the full rule; the [TypeScript SDK](/sdk/streaming) does all of this for you. diff --git a/docs/src/content/docs/pipes.mdx b/docs/src/content/docs/pipes.mdx index 64ce45bd..fa421c26 100644 --- a/docs/src/content/docs/pipes.mdx +++ b/docs/src/content/docs/pipes.mdx @@ -222,7 +222,7 @@ Ship a curated "top pages" endpoint that the public dashboard can call with no t ```bash curl "http://localhost:8080/v1/pipes/top_pages?limit=10" - # → [{"page":"/home","views":1422}, {"page":"/pricing","views":910}, ...] + # → [{"page":"\/home","views":1422}, {"page":"\/pricing","views":910}, ...] ``` ## See also From 417d7b037123739b5a18fb69fc62265cc5723660 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 12:02:39 -0400 Subject: [PATCH 57/70] fix(api): leave / unescaped wherever ClickHouse renders JSON ClickHouse's JSON writer escapes "/" as "\/" by default. The structured-query reader and the /v1/ops/query proxy inherited that from the server, or whatever a tenant's profile chose, and the chtypes export behind the NATS/SSE row wrote it too. Pin output_format_json_escape_forward_slashes=0 on all three, so /v1/query, pipes and the stream keep the "/home" spelling of earlier releases and /v1/ops/query matches them. Measured on ClickHouse 24.8.14.39 and 26.8.15.10 and on the 26.8 chtypes artifact: honoured everywhere a string is written, with no unsupported-setting decline. Cached bytes change, so chRendering moves to JSONEachRow/3. The reader comment no longer claims every byte-changing knob is pinned: a profile can still set, for one, output_format_decimal_trailing_zeros. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- CHANGELOG.md | 2 + docs/src/content/docs/api.md | 11 ++--- docs/src/content/docs/architecture.md | 7 ++-- docs/src/content/docs/getting-started.md | 2 +- docs/src/content/docs/pipes.mdx | 2 +- internal/api/clickhouse_http.go | 27 +++++++----- internal/api/clickhouse_http_test.go | 1 + internal/api/ingest_framing.go | 5 ++- internal/api/ingest_framing_test.go | 8 ++-- internal/api/query.go | 7 +++- internal/api/query_test.go | 5 ++- internal/typelayer/ingest.go | 11 +++-- internal/typelayer/ingest_test.go | 41 ++++++++++++++++--- internal/typelayer/roletable_test.go | 6 +-- tests/integration/query_types_test.go | 23 ++++++++--- .../integration/testdata/query_types_pin.json | 2 +- 17 files changed, 114 insertions(+), 48 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 3e89c49c..5c1324e6 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -72,7 +72,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 16. **Bearer-token-only CORS posture (security)** — Bearer JWT on every request, no cookies/sessions; `corsMiddleware` deliberately **never** emits `Access-Control-Allow-Credentials` (not needed, and `*` + credentials is a spec violation browsers reject). `cors.allowed_origins` (settings directory, per tenant: a tenant route is decorated from the list of the tenant it names, everything else from tenant `0`'s — `corsOrigins`) controls who can *read* responses, not cookie scope; CSRF protection is structural. Don't reintroduce cookie auth or `Allow-Credentials` without a design discussion — answers GitHub #29/#30. Code: `internal/api/router.go`. 17. **Non-fatal boot** — schema-discovery failure on boot is non-fatal: `internal/app` records an `api.BootState`, binds `:8080`, serves 503 on `/livez`/`/readyz` with the diagnostic, and retries via `SchemaRegistry.RetryRefresh` (jittered backoff 2s → 60s), per tenant over a nested directory: `/livez` is 503 while no tenant has completed a first discovery, then sticky 200, and a tenant's outage after that is its log line and counter, never a probe failure. Until a tenant's first discovery its table lookups are a 503 with `Retry-After`, not a 404. Bounds supervisor restart loops. 18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention; `/readyz` pings every open ClickHouse pool at once and is ready at the first answer, 503 naming each when none answers); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. -19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), as RFC 3339 in UTC at the column's scale (e.g. `"2026-06-21T04:00:00.123Z"`), because `date_time_output_format=iso` is pinned on both surfaces — the ingest export settings (`internal/typelayer/ingest.go`) and the read path's `chReadSettingsFixed` (`internal/api/clickhouse_http.go`), which must change together with a `chRendering` bump — so there is no separate WaveHouse rewrite step to keep in sync and live and query reads can't drift on spelling *or* instant (#372). The column's or server's zone affects only how zone-less *input* is read, never the rendered spelling. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. +19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), as RFC 3339 in UTC at the column's scale (e.g. `"2026-06-21T04:00:00.123Z"`), because `date_time_output_format=iso` is pinned on both surfaces (as is `output_format_json_escape_forward_slashes=0`, so neither escapes `/`) — the ingest export settings (`internal/typelayer/ingest.go`) and the read path's `chReadSettingsFixed` (`internal/api/clickhouse_http.go`), which must change together with a `chRendering` bump — so there is no separate WaveHouse rewrite step to keep in sync and live and query reads can't drift on spelling *or* instant (#372). The column's or server's zone affects only how zone-less *input* is read, never the rendered spelling. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. 20. **Sealed MQ boundary** — only `internal/mq` imports NATS/JetStream (`github.com/nats-io/…`), enforced by the `depguard` rule in `.golangci.yml`, so `make lint` fails on a leak in every package it builds (the `integration`-tagged files under `tests/` are outside lint's build context — keep them clean by convention, through `mq.Broker`). A test outside `internal/mq` that needs a real NATS server goes through `internal/mq/natstest`, which stands one up from the shipped `deployments/nats` files and hands back a URL and passwords, never a NATS type. The boundary is semantic as well: everything else addresses events by `mq.Topic` and states intent through mq-owned interfaces (`Publisher`, `Consumer`, `DeadLetterer`, `Purger`, `Replayer`, …), and never builds a subject, names a stream, or reasons in sequences — so a subject, stream, or broker change lands in one package ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 4; story 5's tenant token landed there alone — `Topic.Tenant`, first in every subject). Don't add a raw accessor (`JetStream()`, `NatsConn()`, `GetServer()`) back, and don't hand-build `"ingest."`/`"dlq."` subjects outside `internal/mq` — widen the mq surface with an intent-level method instead. 21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` (with its test helper `typelayertest`) is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503` (generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds every row of a role that has a row `filter` with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and, on Linux, glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. diff --git a/CHANGELOG.md b/CHANGELOG.md index f7c4f793..041088b2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -100,6 +100,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/sql_classify.go` (was `clickhouse_exec.go`, trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`, `internal/settings/validate.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), `Date`/`Date32` are `"2026-06-21"` where the native driver gave `"2026-06-21T00:00:00Z"`, and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. The reader's HTTP connections are capped per connection tuple (URL, user, password, database, TLS) at the largest `max_open_conns` among the tenants sharing it, and its requests carry the tenant's `clickhouse.headers`; the settings warning about a plaintext HTTP hop (`internal/settings/validate.go`) now says it carries the credentials on every query and insert. The 64 MiB cap is new on these two paths, so a pipe (which has no row limit) returning more than 64 MiB now fails; `/v1/query` normally stays under it through its default row cap. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. +- **ClickHouse-rendered JSON leaves `/` unescaped on every path** (`internal/api/{clickhouse_http,query,ingest_framing}.go` (+ tests), `internal/typelayer/ingest.go` (+ tests), `tests/integration/query_types_test.go` (+ its pin), `docs/src/content/docs/{api,architecture,getting-started}.md`, `docs/src/content/docs/pipes.mdx`, `AGENTS.md`): ClickHouse's JSON writer spells `/` as `\/` by default, and a tenant's profile could choose either. `output_format_json_escape_forward_slashes=0` is now pinned on `/v1/query`, pipes, `/v1/ops/query` and the ingest export behind the NATS/SSE row, so `/v1/query`, pipes and the stream keep the `"/home"` spelling of earlier releases, and `/v1/ops/query` changes from ClickHouse's default `"\/home"` to `"/home"` (a `SETTINGS` clause in the SQL sent to `/v1/ops/query` still overrides it). The cache's rendering marker moved to `JSONEachRow/3`, so a deploy starts with a cold query cache. + - **WH001 (no hard-wrapped prose) now applies to every tracked Markdown file, with no carve-out** (`.github/.markdownlint.json` (deleted), `.claude/.markdownlint.json` (deleted), `.claude/skills/integration-astro-view-transitions/` (deleted), `.markdownlint-cli2.jsonc`, `.github/workflows/README.md`, `.claude/skills/pm-triage/references/routine.md`, `AGENTS.md`, `scripts/docs-prose.sh`, `.github/prompts/docs-review.md`, `docs/src/content/docs/claude-code.md`, `docs/src/content/docs/development.md`, `.claude/agents/docs-reviewer.md`): two path-scoped configs had switched WH001 off under `.github/` and `.claude/` ever since [#489](https://github.com/Wave-RF/WaveHouse/pull/489) introduced the rule — baked in from the start rather than added in response to a discovered problem — which left the repo documenting the rule three ways and disagreeing with itself: `CONTRIBUTING.md` promises contributors `make lint` enforces it *everywhere*, while `AGENTS.md` and the `.markdownlint-cli2.jsonc` header wrote up the carve-out. Not theoretical: on [#520](https://github.com/Wave-RF/WaveHouse/pull/520) a reviewer correctly flagged a hard-wrapped bullet in `.github/workflows/README.md`, an agent pointed at `"WH001": false` for that path and pushed back, and the reviewer recorded a *learning* never to flag WH001 there — the wrong invariant, learned off the wrong side of the contradiction ([#521](https://github.com/Wave-RF/WaveHouse/issues/521)). Both configs are deleted — each held nothing but the override, so the root `.markdownlint.json` governs again — and the 51 hard-wrapped paragraphs they were hiding are joined: 41 in `.github/workflows/README.md` and 10 in `.claude/skills/pm-triage/references/routine.md`, mechanical joins with no wording changed and every fenced block, table row, and heading byte-identical either side of the reflow. Deleted with them: the wizard-installed PostHog skill at `.claude/skills/integration-astro-view-transitions/` — 9 files, ~1,456 lines, including an 809-line `EXAMPLE.md` copied wholesale from `PostHog/context-mill`. Its integration job finished in [#277](https://github.com/Wave-RF/WaveHouse/pull/277), nothing in the repo calls it, and the docs-site setup it once described is documented where it belongs — in `docs/src/components/PostHog.astro` and this file. Keeping unowned third-party prose in the tree means content that drifts silently on every upstream bump and that nobody here reviews; it was also the single file that would have needed a special-case lint exclusion, so removing it is what lets WH001 apply with **no exception at all** rather than one documented one. Its two inventory rows in `claude-code.md` go with it, as does the now-dead `docs/posthog-setup-report.md` entry in the `scripts/docs-prose.sh` denylist (the wizard's other artifact, deleted back in [#502](https://github.com/Wave-RF/WaveHouse/pull/502)) and the copies of that denylist in `AGENTS.md` and `.github/prompts/docs-review.md`, which the script's header requires be kept in lockstep. Review of the change then turned up four more things the exclusion had been hiding, all fixed here: **WH001 has a blind spot** — `no-hard-wrapped-prose.mjs` classifies any line indented four or more spaces as an indented code block, so a *nested* list item is never joined, which left three hard-wrapped bullets in `.github/workflows/README.md` §"Adding a job" that the autofix could not see (unwrapped by hand; they were the last hard-wrapped prose paragraphs in the repo) and made `AGENTS.md`'s and `development.md`'s "a list item is joined as a unit" wrong for nested items (both now state the four-space caveat); the `scripts/docs-prose.sh` header told readers to keep its denylist in lockstep with **two** sibling copies when there are **three** — the missed one being `.claude/agents/docs-reviewer.md`, the gating subagent's own system prompt, which had in fact been silently out of sync for the whole life of the `posthog-setup-report.md` exclusion; the `.markdownlint-cli2.jsonc` header's "applies to every tracked Markdown file" was exact for WH001 but not WH002, which returns early on anything that isn't `.mdx`; and the job-graph diagram omitted `docs-deploy`'s `needs` edges from `unit`, `integration`, and `e2e`, contradicting invariant 2 three lines below it. The denylist also drops its `PERF-CLAIMS-REVIEW.md` entry — unlike the wizard artifact this one names a file that was **never tracked** at all, so it guarded a hypothetical; the list's other general cases are patterns (`*.draft.md`, `*.old.md`) that already cover a one-off review document, and a literal filename restated in four places is the outlier. `scripts/docs-prose.sh all` still resolves the same 27-file prose set. ### Removed diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index efa35bd8..2d2a9d9e 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -433,7 +433,7 @@ A batch aborted partway — a `503` or `500` after some leading records were alr ### `POST /v1/ops/query` — Query ClickHouse -Executes a SQL statement directly against ClickHouse. **WaveHouse proxies the SQL string verbatim to ClickHouse's HTTP interface** — any statement ClickHouse accepts works, including arbitrary DDL/DML/SYSTEM verbs and inline FORMAT directives. One statement per request: ClickHouse's HTTP interface refuses multi-statement input (`SELECT 1; TRUNCATE t` is `Code: 62 … Multi-statements are not allowed`, answered `400 clickhouse.rejected`), so send each statement as its own request. Read queries return a JSON array of result rows; mutations/DDL return HTTP 200 with `[]` on success. The proxy does not ask ClickHouse to hold its answer until the statement ends, so a statement that fails after ClickHouse began streaming a large result comes back `200` with ClickHouse's partial output followed by its exception text, which is not valid JSON: treat a body that does not parse as a failure. The request body is capped at 16 MiB. DateTime columns are ISO-8601 formatted via the upstream `date_time_output_format=iso` setting — the same server-side rendering `/v1/query`, pipes and the stream use, so a `DateTime64(3)` whole-second value returns `.000Z` on all of them; other types are returned as ClickHouse renders them under `FORMAT JSON`. +Executes a SQL statement directly against ClickHouse. **WaveHouse proxies the SQL string verbatim to ClickHouse's HTTP interface** — any statement ClickHouse accepts works, including arbitrary DDL/DML/SYSTEM verbs and inline FORMAT directives. One statement per request: ClickHouse's HTTP interface refuses multi-statement input (`SELECT 1; TRUNCATE t` is `Code: 62 … Multi-statements are not allowed`, answered `400 clickhouse.rejected`), so send each statement as its own request. Read queries return a JSON array of result rows; mutations/DDL return HTTP 200 with `[]` on success. The proxy does not ask ClickHouse to hold its answer until the statement ends, so a statement that fails after ClickHouse began streaming a large result comes back `200` with ClickHouse's partial output followed by its exception text, which is not valid JSON: treat a body that does not parse as a failure. The request body is capped at 16 MiB. DateTime columns are ISO-8601 formatted via the upstream `date_time_output_format=iso` setting — the same server-side rendering `/v1/query`, pipes and the stream use, so a `DateTime64(3)` whole-second value returns `.000Z` on all of them — and `/` is left unescaped (`output_format_json_escape_forward_slashes=0`, as on those paths) unless the statement's own `SETTINGS` clause says otherwise; other types are returned as ClickHouse renders them under `FORMAT JSON`. :::note[Inline `FORMAT` overrides the JSON envelope] ClickHouse's inline `FORMAT` clause (e.g. `SELECT 1 FORMAT CSV` or `… FORMAT Pretty`) takes precedence over the URL-level `default_format=JSON` setting. When the SQL contains an explicit `FORMAT`, the proxy forwards ClickHouse's raw response body (CSV, Pretty, TSV, …) and passes through the upstream `Content-Type` header — `text/csv`, `text/tab-separated-values`, etc. — so consumers see the right MIME type. The "extract the `data` array" behavior only applies when ClickHouse returned the `FORMAT JSON` envelope, which is the default. @@ -466,7 +466,7 @@ An optional `?tenant=` names the [tenant](/deployment#the-nested-settings-di | `sql` | string | Yes | SQL forwarded verbatim to ClickHouse's HTTP interface. | :::note[No parameter binding on this endpoint (yet)] -The earlier handler accepted a `params` array bound to `?` placeholders; the HTTP proxy doesn't. ClickHouse's native named-param syntax (`WHERE id = {id:UInt32}` with `param_id=42` on the URL query string) is *not* forwarded today either — the proxy only sets `default_format`, `date_time_output_format`, and `database` on the upstream URL, and the request body is `{"sql": "..."}` with no escape hatch for query-string params. The current contract is "send raw SQL, get rows back": inline literals into the SQL for now. For safe binding from user-supplied input, use the structured query endpoint (`POST /v1/query?table={table}`) — that's its job. +The earlier handler accepted a `params` array bound to `?` placeholders; the HTTP proxy doesn't. ClickHouse's native named-param syntax (`WHERE id = {id:UInt32}` with `param_id=42` on the URL query string) is *not* forwarded today either — the proxy only sets `default_format`, `date_time_output_format`, `output_format_json_escape_forward_slashes`, and `database` on the upstream URL, and the request body is `{"sql": "..."}` with no escape hatch for query-string params. The current contract is "send raw SQL, get rows back": inline literals into the SQL for now. For safe binding from user-supplied input, use the structured query endpoint (`POST /v1/query?table={table}`) — that's its job. ::: **Response:** @@ -575,6 +575,7 @@ JSON array of result rows, **rendered by ClickHouse**: the query runs over its H | `DateTime`, `DateTime64` | `"2026-06-21T04:00:00.123Z"` — RFC 3339 in UTC whatever zone the column declares; byte-identical to the [SSE stream](#get-v1stream--server-sent-events-stream) for the same row (see [Timestamp rendering](#timestamp-rendering)) | | `Decimal*` | a JSON **number** (`12.5`), not a string | | `Int64`/`UInt64` past 2^53 | an unquoted number — still lossy in a JavaScript `number`; read it as text if you need every digit | +| `String` | `/` as `/` (`"/home"`): WaveHouse pins `output_format_json_escape_forward_slashes=0`, where ClickHouse's default writes `"\/home"` | | `FixedString(n)` | a string padded to `n` bytes with `\u0000` | | `Enum*` | the name, not the ordinal | | `Nullable(T)` | `null` for a SQL `NULL` | @@ -674,13 +675,13 @@ event: schema data: {"table_name":"clicks","columns":["page","button","score","received_timestamp"]} id: 2026-03-24T12:00:00.123Z -data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:00.123Z","row":["\/home","signup",42.5,"2026-03-24T11:59:58.512Z"]} +data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:00.123Z","row":["/home","signup",42.5,"2026-03-24T11:59:58.512Z"]} id: 2026-03-24T12:00:01.456Z -data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["\/pricing","cta",7,"2026-03-24T12:00:01.456Z"]} +data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,"2026-03-24T12:00:01.456Z"]} ``` -String values are ClickHouse's JSON rendering, which escapes `/` as `\/` — valid JSON that any parser reads back as `/`. +A `/` in a string arrives as `/`, not as the `\/` ClickHouse's JSON writer produces by default: WaveHouse pins `output_format_json_escape_forward_slashes=0` on the export that produces these rows and on `/v1/query`, pipes and `/v1/ops/query`. A raw consumer must keep the most recent announced column list and zip each `row` against it; a column the record omitted still has its slot, holding its evaluated `DEFAULT` (or the type's default — `null` only on a `Nullable` column with none), so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you: `.stream()` and `.liveQuery()` zip each row into an object. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index d10b0e24..452c44ef 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -86,8 +86,8 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. - **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (brackets that do not balance are a whole-request `400 invalid json: unterminated json array`, anything but whitespace after the closing `]` a whole-request `400 invalid json: content after the closing ']' of the json array`, and a `…WithNames` header naming an unknown or repeated column a whole-request `400 clickhouse.rejected` with code 117, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read, without affecting other tenants. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's duplicate window (two minutes on the embedded broker) — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). -- **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso`, the same spelling the structured-query path and the SSE wire use, so a timestamp reads the same on every surface. -- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `output_format_json_named_tuples_as_objects=1`, `date_time_output_format=iso`, so a timestamp is RFC 3339 in UTC and matches the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so `chconn.Classify` and `writeCHError` apply as on every other ClickHouse path, and a response past the 64 MiB cap is `clickhouse.response_too_large`. The read paths hold at most the tenant's pool size in HTTP connections — its `max_open_conns`, or the largest among the tenants sharing its connection tuple (address, database, user, password, TLS) — one cap per tuple, like the native pool; tenants on one server with a different database, user, password or TLS settings each have their own. +- **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` and `/` left unescaped via `output_format_json_escape_forward_slashes=0`, the same spellings the structured-query path and the SSE wire use, so a timestamp and a `/` read the same on every surface. +- **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `output_format_json_named_tuples_as_objects=1`, `output_format_json_escape_forward_slashes=0`, `date_time_output_format=iso`, so a `/` is not escaped and a timestamp is RFC 3339 in UTC, both as on the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so `chconn.Classify` and `writeCHError` apply as on every other ClickHouse path, and a response past the 64 MiB cap is `clickhouse.response_too_large`. The read paths hold at most the tenant's pool size in HTTP connections — its `max_open_conns`, or the largest among the tenants sharing its connection tuple (address, database, user, password, TLS) — one cap per tuple, like the native pool; tenants on one server with a different database, user, password or TLS settings each have their own. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) chconn.Target` beside it — each resolves the request's tenant per call, and the zero target (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` before a cached result is served or a query runs. The cached paths resolve it after their cache `Lookup`, so the snapshot predates the connection (see `cache.go` below). - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for one tenant's per-table parked counts (optionally one table) and its total — the tenant `?tenant=` names, read strictly by `opsTenant`, tenant `0` without it. The tenant is looked up in the MQ, not the settings registry, so a rejected or removed tenant's parked rows are read like a served one's; a tenant with no dead-letter queue (`mq.ErrNoDeadLetterQueue`) is a 404, and any other failure to read it a 500. The queue itself is `internal/mq`'s. @@ -376,7 +376,8 @@ Client POST /v1/ops/query → Decode {"sql": "..."} from the request body. → POST the SQL verbatim to ClickHouse's HTTP interface at ://:/?default_format=JSON - &date_time_output_format=iso&database= + &date_time_output_format=iso + &output_format_json_escape_forward_slashes=0&database= Auth via X-ClickHouse-User / X-ClickHouse-Key headers. Bound by a clickhouse.query_timeout context derived from the inbound request — client disconnect cancels the upstream call. diff --git a/docs/src/content/docs/getting-started.md b/docs/src/content/docs/getting-started.md index 16425029..1e231737 100644 --- a/docs/src/content/docs/getting-started.md +++ b/docs/src/content/docs/getting-started.md @@ -95,7 +95,7 @@ curl -N "http://localhost:8080/v1/stream?table=clicks" curl -N "http://localhost:8080/v1/stream?table=clicks&since=2026-03-24T11:00:00Z" ``` -Rows arrive **positionally**, so raw `curl` output looks like `"row":["\/home","signup",42.5,"2026-03-24T11:59:58.512Z"]` rather than named fields (strings are ClickHouse's JSON rendering, which escapes `/` as `\/`; any JSON parser reads it back as `/`). The stream sends an `event: schema` frame before the first row and again when the column list changes, and a raw client must pair each row against the **most recent** frame rather than the first. +Rows arrive **positionally**, so raw `curl` output looks like `"row":["/home","signup",42.5,"2026-03-24T11:59:58.512Z"]` rather than named fields. The stream sends an `event: schema` frame before the first row and again when the column list changes, and a raw client must pair each row against the **most recent** frame rather than the first. That re-announcement is not guaranteed in one case: after a gap-fill across a column change, live rows can arrive without a fresh frame ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). Drop a row whose length disagrees with the last announced list rather than zipping it, and reconnect to resynchronize — an arity check cannot see a same-length change such as a `RENAME COLUMN`. See [the wire format](/api#get-v1stream--server-sent-events-stream) for the frame sequence and the full rule; the [TypeScript SDK](/sdk/streaming) does all of this for you. diff --git a/docs/src/content/docs/pipes.mdx b/docs/src/content/docs/pipes.mdx index fa421c26..64ce45bd 100644 --- a/docs/src/content/docs/pipes.mdx +++ b/docs/src/content/docs/pipes.mdx @@ -222,7 +222,7 @@ Ship a curated "top pages" endpoint that the public dashboard can call with no t ```bash curl "http://localhost:8080/v1/pipes/top_pages?limit=10" - # → [{"page":"\/home","views":1422}, {"page":"\/pricing","views":910}, ...] + # → [{"page":"/home","views":1422}, {"page":"/pricing","views":910}, ...] ``` ## See also diff --git a/internal/api/clickhouse_http.go b/internal/api/clickhouse_http.go index 95f8f2a4..d9fbf605 100644 --- a/internal/api/clickhouse_http.go +++ b/internal/api/clickhouse_http.go @@ -22,22 +22,26 @@ import ( // every cached read's key (queryCacheKey), so two builds that render rows // differently never serve each other's entries from a shared cache during a // rolling deploy. Change it with any change to the rendering settings below. -const chRendering = "JSONEachRow/2" +const chRendering = "JSONEachRow/3" // chReadSettingsFixed go on every request the cached read paths send, ahead // of the role's caps. // // Rendering: ClickHouse renders the rows, so a Decimal's spelling, a // DateTime64's scale, an Enum's name and an IPv6's compression are the -// server's own. Every knob that changes the bytes is pinned rather than -// inherited, because a tenant's server or profile may set any of them: -// 64-bit integers and decimals as bare numbers, NaN and Inf as null, a named -// tuple as an object, and DateTime as RFC 3339 in UTC, -// `YYYY-MM-DDThh:mm:ss[.fff]Z` with the column's scale, whatever the column's -// or the server's zone — the spelling the SSE wire carries (typelayer's -// export pins the same), so neither a client nor the cache needs to know the -// server's zone (#372). Date and Date32 are unaffected. Measured on -// 26.8.15.10. +// server's own. These knobs are pinned rather than inherited, because a +// tenant's server or profile may set any of them: 64-bit integers and +// decimals as bare numbers, NaN and Inf as null, a named tuple as an object, +// `/` left unescaped (`"/home"`, where ClickHouse's default writes +// `"\/home"`), and DateTime as RFC 3339 in UTC, `YYYY-MM-DDThh:mm:ss[.fff]Z` +// with the column's scale, whatever the column's or the server's zone. The +// SSE wire spells `/` and DateTime the same way (typelayer's export pins +// both), so neither a client nor the cache needs to know the server's zone +// (#372). Date and Date32 are unaffected. Measured on 26.8.15.10. A profile +// can still change the bytes through a knob not pinned here, such as +// output_format_decimal_trailing_zeros, output_format_json_array_of_rows or +// output_format_trim_fixed_string (measured on 26.8.15.10; the last is +// unknown to 24.8, where sending it would fail every read). // // Failure: wait_end_of_query buffers the result server-side until the query // has finished, and http_write_exception_in_output_format=0 keeps an @@ -52,7 +56,7 @@ const chRendering = "JSONEachRow/2" // rather than letting it run on after nobody waits for it. // // Every one of these is known to ClickHouse 24.8 and later (measured on -// 24.8.14.39 and 26.6.3.62). +// 24.8.14.39 and 26.8.15.10). var chReadSettingsFixed = map[string]string{ "default_format": "JSONEachRow", "wait_end_of_query": "1", @@ -61,6 +65,7 @@ var chReadSettingsFixed = map[string]string{ "output_format_json_quote_64bit_integers": "0", "output_format_json_quote_decimals": "0", "output_format_json_quote_denormals": "0", + "output_format_json_escape_forward_slashes": "0", "date_time_output_format": "iso", "output_format_json_named_tuples_as_objects": "1", } diff --git a/internal/api/clickhouse_http_test.go b/internal/api/clickhouse_http_test.go index f7f9a73a..8c6e0575 100644 --- a/internal/api/clickhouse_http_test.go +++ b/internal/api/clickhouse_http_test.go @@ -258,6 +258,7 @@ func TestCHReader_Request(t *testing.T) { // Pinned here rather than read from the map: the SSE wire (typelayer's // export) and the SDK compare against this spelling. assert.Equal(t, "iso", got.query.Get("date_time_output_format"), "DateTime as RFC 3339 in UTC") + assert.Equal(t, "0", got.query.Get("output_format_json_escape_forward_slashes"), `"/home", not ClickHouse's default "\/home"`) assert.Equal(t, "2", got.query.Get("readonly")) assert.Equal(t, "warehouse", got.query.Get("database")) assert.Equal(t, []string{"/home", `a\tb`}, ch.params()) diff --git a/internal/api/ingest_framing.go b/internal/api/ingest_framing.go index eb9b4442..eac8ee17 100644 --- a/internal/api/ingest_framing.go +++ b/internal/api/ingest_framing.go @@ -208,8 +208,9 @@ func eventIDAt(line []byte, idx int) (string, bool) { } if cell[0] == '"' { // One scalar string, not the record: the cell is JSON-encoded by - // ClickHouse's own writer (it escapes "/" as "\/"), so Go's own - // string-literal unquoting would refuse it. + // ClickHouse's own writer, so it is JSON-decoded. Go's own + // string-literal unquoting reads a different escape grammar (it + // refuses JSON's "\/", for one). var s string if err := json.Unmarshal(cell, &s); err != nil || s == "" { return "", false diff --git a/internal/api/ingest_framing_test.go b/internal/api/ingest_framing_test.go index 01d827d4..f251c95b 100644 --- a/internal/api/ingest_framing_test.go +++ b/internal/api/ingest_framing_test.go @@ -152,9 +152,9 @@ func TestCellAt(t *testing.T) { assert.False(t, ok, "an unterminated row yields nothing") } -// TestEventIDAt: the id is the STORED value, a string cell is JSON-decoded -// because ClickHouse's writer escapes "/" as "\/" — Go's own string-literal -// unquoting refuses that — and a null cell is no id at all. +// TestEventIDAt: the id is the STORED value, a string cell is JSON-decoded — +// JSON's escapes, "\/" included, which Go's own string-literal unquoting +// refuses — and a null cell is no id at all. func TestEventIDAt(t *testing.T) { t.Parallel() for _, tt := range []struct { @@ -164,7 +164,7 @@ func TestEventIDAt(t *testing.T) { want string ok bool }{ - {"a quoted string is decoded", `["\/a", "evt-1", 0]`, 1, "evt-1", true}, + {"a quoted string is decoded", `["/a", "evt-1", 0]`, 1, "evt-1", true}, {"a slash escape survives", `["\/a\/b", 0]`, 0, "/a/b", true}, {"a number is its digits", `["x", 18446744073709551615]`, 1, "18446744073709551615", true}, {"an empty string is no id", `["x", ""]`, 1, "", false}, diff --git a/internal/api/query.go b/internal/api/query.go index e1b675a1..6e6e6f87 100644 --- a/internal/api/query.go +++ b/internal/api/query.go @@ -48,7 +48,8 @@ type QueryHandler struct { // target resolves the tenant's ClickHouse HTTP wiring per request // (chconn.Pools.Target in production): the base URL (e.g. // `http://localhost:8123`) the handler appends query-string params - // (`default_format`, `database`, `date_time_output_format`) to and POSTs + // (`default_format`, `database`, `date_time_output_format`, + // `output_format_json_escape_forward_slashes`) to and POSTs // the SQL against, plus the credentials and database; the zero Target is // a tenant on no pool, a 503. queryTimeout bounds each proxied query. // Funcs, not values, so a settings reload applies to the next request. @@ -226,6 +227,10 @@ func (h *QueryHandler) Handle(w http.ResponseWriter, r *http.Request) { // re-parse ClickHouse's default `YYYY-MM-DD HH:MM:SS`. Matches the prior // handler's RFC3339Nano output close enough for downstream callers. q.Set("date_time_output_format", "iso") + // `/` as `/`, as /v1/query and pipes spell it (chReadSettingsFixed), not + // ClickHouse's default `\/`. A SETTINGS clause in the SQL still wins over + // the URL (measured on 26.8.15.10). + q.Set("output_format_json_escape_forward_slashes", "0") if target.Database != "" { q.Set("database", target.Database) } diff --git a/internal/api/query_test.go b/internal/api/query_test.go index 30b41933..8cf4cbf5 100644 --- a/internal/api/query_test.go +++ b/internal/api/query_test.go @@ -160,7 +160,9 @@ func TestQueryHandler_NilHTTPClientReturnsError(t *testing.T) { // TestQueryHandler_ForwardsSQLToClickHouse pins the proxy contract: // - Request SQL is sent as the HTTP body verbatim. // - default_format=JSON and date_time_output_format=iso are set so the -// response envelope is predictable. +// response envelope is predictable, and +// output_format_json_escape_forward_slashes=0 so `/` is spelled as on +// /v1/query. // - Database from constructor lands in the query string. // - X-ClickHouse-User / X-ClickHouse-Key headers carry the credentials. // @@ -203,6 +205,7 @@ func TestQueryHandler_ForwardsSQLToClickHouse(t *testing.T) { assert.Equal(t, "/", gotPath) assert.Contains(t, gotQuery, "default_format=JSON") assert.Contains(t, gotQuery, "date_time_output_format=iso") + assert.Contains(t, gotQuery, "output_format_json_escape_forward_slashes=0") assert.Contains(t, gotQuery, "database=default") assert.Equal(t, sql, gotBody, "SQL body must be forwarded verbatim, no parsing") assert.Equal(t, "default", gotUser, "username must be sent via X-ClickHouse-User") diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go index d79b5583..047eb96e 100644 --- a/internal/typelayer/ingest.go +++ b/internal/typelayer/ingest.go @@ -35,13 +35,16 @@ type IngestOptions struct { // inserts JSONCompactEachRow, so the detect_header settings have no real-INSERT // twin to keep in step. ok is false for a format Ingest does not parse. // -// The export renders DateTime as RFC 3339 in UTC (`…T…Z`, the column's scale), -// the spelling the query paths pin too, so a published row reads the same as a -// queried one and carries its instant whatever the zone; the worker's -// best_effort INSERT stores that exact instant. Measured on 26.8.15.10. +// The export renders DateTime as RFC 3339 in UTC (`…T…Z`, the column's scale) +// and leaves `/` unescaped (`"/home"`, where the writer's default is +// `"\/home"`), the spellings the query paths pin too, so a published row spells +// a timestamp and a `/` as a queried one does, and carries its instant whatever +// the zone; the worker's best_effort INSERT stores that exact instant. Measured +// on 26.8.15.10. func parseSettings(format Format, opts IngestOptions) (settings map[string]string, ok bool) { settings = InsertSettings() settings["date_time_output_format"] = "iso" + settings["output_format_json_escape_forward_slashes"] = "0" switch format { case FormatJSONEachRow, FormatCSVWithNames, FormatTSVWithNames: case FormatCSV: diff --git a/internal/typelayer/ingest_test.go b/internal/typelayer/ingest_test.go index 34fefbbf..1887a98f 100644 --- a/internal/typelayer/ingest_test.go +++ b/internal/typelayer/ingest_test.go @@ -164,11 +164,11 @@ func TestIngest_EphemeralInputFollowsTheFormat(t *testing.T) { body string want string }{ - {"JSONEachRow", FormatJSONEachRow, `{"page":"/a","ip":"1.2.3.4"}` + "\n", `["\/a", 7]`}, - {"CSVWithNames", FormatCSVWithNames, "page,ip\n/a,1.2.3.4\n", `["\/a", 7]`}, - {"TSVWithNames", FormatTSVWithNames, "ip\tpage\n1.2.3.4\t/a\n", `["\/a", 7]`}, - {"CSV is the wire columns", FormatCSV, "/a,3\n", `["\/a", 3]`}, - {"TSV is the wire columns", FormatTSV, "/a\t3\n", `["\/a", 3]`}, + {"JSONEachRow", FormatJSONEachRow, `{"page":"/a","ip":"1.2.3.4"}` + "\n", `["/a", 7]`}, + {"CSVWithNames", FormatCSVWithNames, "page,ip\n/a,1.2.3.4\n", `["/a", 7]`}, + {"TSVWithNames", FormatTSVWithNames, "ip\tpage\n1.2.3.4\t/a\n", `["/a", 7]`}, + {"CSV is the wire columns", FormatCSV, "/a,3\n", `["/a", 3]`}, + {"TSV is the wire columns", FormatTSV, "/a\t3\n", `["/a", 3]`}, } { t.Run(tc.name, func(t *testing.T) { batch, err := tbl.Ingest(tc.format, []byte(tc.body)) @@ -289,6 +289,37 @@ func TestIngest_DateTimeExportsAsRFC3339UTC(t *testing.T) { `"2026-03-24T12:00:00Z", "2026-03-24T12:00:00.120Z", "2026-03-24", "1960-01-02"]`, string(batch.Rows[0].Line)) } +// TestIngest_ForwardSlashExportsUnescaped: the exported row leaves `/` as is +// wherever the writer puts a string — a value, an array element, a map key, a +// named tuple's field name — where the writer's default is `\/`, as the query +// paths render it. An insert check on the same parse changes nothing. +func TestIngest_ForwardSlashExportsUnescaped(t *testing.T) { + eng := testEngine(t, &discovery.TableSchema{Name: "paths", Columns: []discovery.Column{ + {Name: "s", Type: "String", Position: 1}, + {Name: "lc", Type: "LowCardinality(String)", Position: 2}, + {Name: "arr", Type: "Array(String)", Position: 3}, + {Name: "m", Type: "Map(String, String)", Position: 4}, + {Name: "t", Type: "Tuple(`n/m` String)", Position: 5}, + }}) + tbl, err := eng.Table(tenant.Default, "paths") + require.NoError(t, err) + t.Cleanup(tbl.Release) + + body := []byte(`{"s":"/home","lc":"a/b","arr":["c/d"],"m":{"k/1":"v/2"},"t":{"n/m":"x/y"}}` + "\n") + const want = `["/home", "a/b", ["c/d"], {"k/1":"v/2"}, {"n/m":"x/y"}]` + for name, checks := range map[string][]Predicate{ + "no checks": nil, + "with check": {{Column: "s", Op: "=", Values: []string{"/home"}}}, + } { + batch, err := tbl.Ingest(FormatJSONEachRow, body, checks...) + require.NoError(t, err, name) + require.Len(t, batch.Rows, 1, name) + require.True(t, batch.Rows[0].Accepted, "%s: %s", name, batch.Rows[0].Message) + require.Empty(t, batch.Rows[0].CheckReason, name) + assert.Equal(t, want, string(batch.Rows[0].Line), name) + } +} + func TestInsertSettings_ReturnsAFreshMap(t *testing.T) { t.Parallel() a := InsertSettings() diff --git a/internal/typelayer/roletable_test.go b/internal/typelayer/roletable_test.go index 28a6d128..ce5f2e63 100644 --- a/internal/typelayer/roletable_test.go +++ b/internal/typelayer/roletable_test.go @@ -194,7 +194,7 @@ func TestRoleTable_DeniedColumnAnExpressionReadsStillCompiles(t *testing.T) { require.NoError(t, err) require.Len(t, batch.Rows, 3) require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) - assert.Equal(t, `["\/home", 7, 0]`, string(batch.Rows[0].Line), "length('0.0.0.0'), and n at its type default NULL") + assert.Equal(t, `["/home", 7, 0]`, string(batch.Rows[0].Line), "length('0.0.0.0'), and n at its type default NULL") assert.Equal(t, 117, batch.Rows[1].Code) assert.Contains(t, batch.Rows[1].Message, "ip") assert.Equal(t, 117, batch.Rows[2].Code) @@ -252,7 +252,7 @@ func TestRoleTable_EphemeralFollowsTheRoleColumns(t *testing.T) { allowed.Release() require.NoError(t, err) require.True(t, batch.Rows[0].Accepted, batch.Rows[0].Message) - assert.Equal(t, `["\/a", 7]`, string(batch.Rows[0].Line)) + assert.Equal(t, `["/a", 7]`, string(batch.Rows[0].Line)) denied, err := eng.RoleTable(tenant.Default, "eph", RoleShape{Columns: []string{"page", "ip_len"}}) require.NoError(t, err) @@ -261,7 +261,7 @@ func TestRoleTable_EphemeralFollowsTheRoleColumns(t *testing.T) { require.NoError(t, err) assert.Equal(t, 117, batch.Rows[0].Code) require.True(t, batch.Rows[1].Accepted, batch.Rows[1].Message) - assert.Equal(t, `["\/b", 0]`, string(batch.Rows[1].Line)) + assert.Equal(t, `["/b", 0]`, string(batch.Rows[1].Line)) } // TestRoleTable_ContradictoryShapeIsAnError: a default for a column the shape diff --git a/tests/integration/query_types_test.go b/tests/integration/query_types_test.go index a7aeda82..fc8c83af 100644 --- a/tests/integration/query_types_test.go +++ b/tests/integration/query_types_test.go @@ -44,13 +44,14 @@ const queryTypesDDL = ` // reach ClickHouse without passing through a driver's own type mapping — // the pin must describe ClickHouse's storage, not clickhouse-go's encoder. // i64/u64 sit past 2^53 so the pin also records how 64-bit integers are -// spelled; fs is shorter than its FixedString(4) so the NUL padding shows. +// spelled; s carries a `/` so it records that one is not escaped; fs is +// shorter than its FixedString(4) so the NUL padding shows. const queryTypesRow = `( -8, -16, -32, -9007199254740993, 8, 16, 32, 18446744073709551615, 0.1, 0.1, 12.50, 1.500, - 'hello', 'lc', 'ab', + 'hello/world', 'lc', 'ab', toUUID('11111111-2222-3333-4444-555555555555'), 'a', true, '10.0.0.1', '::1', '2026-01-15', '2026-01-15 10:30:00', '2026-01-15 10:30:00.123', @@ -68,9 +69,10 @@ const queryTypesRow = `( // change and belongs in the CHANGELOG. // // ClickHouse renders the body (FORMAT JSONEachRow under the reader's pinned -// output settings): keys in SELECT order, Decimal as a JSON number, DateTime -// as RFC 3339 in UTC at the column's scale (date_time_output_format=iso). -// Measured on 26.8.15.10. +// output settings): keys in SELECT order, Decimal as a JSON number, `/` +// unescaped (output_format_json_escape_forward_slashes=0), DateTime as RFC +// 3339 in UTC at the column's scale (date_time_output_format=iso). Measured +// on 26.8.15.10. func TestQuery_TypeRendering_Pin(t *testing.T) { e := env(t) @@ -101,3 +103,14 @@ func TestQuery_TypeRendering_Pin(t *testing.T) { "the /v1/query type-rendering contract changed — re-record deliberately "+ "with WAVEHOUSE_UPDATE_PIN=1 and document the diff") } + +// TestOpsQuery_SlashRendering: the raw SQL proxy spells `/` as /v1/query +// does, not as ClickHouse's default `\/`. +func TestOpsQuery_SlashRendering(t *testing.T) { + e := env(t) + + got := postJSON(t, e.baseURL+"/v1/ops/query", `{"sql":"SELECT '/home' AS p"}`) + require.Equal(t, http.StatusOK, got.status, "body: %s", got.raw) + require.Contains(t, got.raw, `"/home"`) + require.NotContains(t, got.raw, `\/`) +} diff --git a/tests/integration/testdata/query_types_pin.json b/tests/integration/testdata/query_types_pin.json index ea3ee9f1..9f7af9e6 100644 --- a/tests/integration/testdata/query_types_pin.json +++ b/tests/integration/testdata/query_types_pin.json @@ -1 +1 @@ -[{"i8":-8,"i16":-16,"i32":-32,"i64":-9007199254740993,"u8":8,"u16":16,"u32":32,"u64":18446744073709551615,"f32":0.1,"f64":0.1,"dec":12.5,"dec64":1.5,"s":"hello","ls":"lc","fs":"ab\u0000\u0000","uu":"11111111-2222-3333-4444-555555555555","en":"a","bl":true,"ip4":"10.0.0.1","ip6":"::1","d":"2026-01-15","dt":"2026-01-15T10:30:00Z","dt64":"2026-01-15T10:30:00.123Z","arr":["a","b"],"m":{"k":7},"nn":42,"nnull":null}] +[{"i8":-8,"i16":-16,"i32":-32,"i64":-9007199254740993,"u8":8,"u16":16,"u32":32,"u64":18446744073709551615,"f32":0.1,"f64":0.1,"dec":12.5,"dec64":1.5,"s":"hello/world","ls":"lc","fs":"ab\u0000\u0000","uu":"11111111-2222-3333-4444-555555555555","en":"a","bl":true,"ip4":"10.0.0.1","ip6":"::1","d":"2026-01-15","dt":"2026-01-15T10:30:00Z","dt64":"2026-01-15T10:30:00.123Z","arr":["a","b"],"m":{"k":7},"nn":42,"nnull":null}] From e21906619b9a8f9bf54a26b094bd84bbfe9aa478 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 12:03:14 -0400 Subject: [PATCH 58/70] test(stream): pin that an unreadable row is declined, not an error A row the type layer cannot read never fails Prepare: the parse succeeds, the filter does not answer for the row, and the withhold is labelled decline. Measured through the production evaluator on the 26.8 artifact for a value the column cannot read, a short row, a truncated array, an object and non-JSON. The new test pins that, and the classifyPrepare case no longer feeds in an error named "unparseable row", which suggested the opposite. The ReasonError and RowWithheld comments now also name error's other source, the type layer refusing a parse call as a whole, and the access control page says a row the engine cannot read is a decline. Co-Authored-By: Claude Opus 5.5 --- docs/src/content/docs/access-control.mdx | 2 +- internal/stream/metrics.go | 8 +++--- internal/stream/roweval.go | 12 ++++++--- internal/stream/roweval_test.go | 32 +++++++++++++++++++++++- internal/typelayer/filter_test.go | 4 +-- 5 files changed, 46 insertions(+), 12 deletions(-) diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index b6024d89..8da838d8 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -367,7 +367,7 @@ The same policy drives every data path, but not every field is meaningful on eve :::caution[Live streams enforce column and row policy, but not resource limits] SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause on a server running that line's default settings, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72), or the parse of the event failed outright; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72), or the parse of the event failed outright; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. diff --git a/internal/stream/metrics.go b/internal/stream/metrics.go index 9c4ed3b6..68d1c1b6 100644 --- a/internal/stream/metrics.go +++ b/internal/stream/metrics.go @@ -95,10 +95,10 @@ func (m *Metrics) FrameDropped(kind string) { // type layer has no compiled schema for the tenant's table, so every // row-filtered subscriber is dark until it does; `drift` means events arrive // under a column list the table's current generation cannot read; `error` means -// ClickHouse raised evaluating the predicate over the row; and `decline` means -// no verdict was reached — a filter that does not compile, a row that does not -// parse, or a filter on a column the inserting role did not write. Only -// `filter` is a policy decision. +// ClickHouse raised evaluating the predicate over the row, or the type layer +// refused to parse the event at all; and `decline` means no verdict was reached +// — a filter that does not compile, a row that does not parse, or a filter on a +// column the inserting role did not write. Only `filter` is a policy decision. func (m *Metrics) RowWithheld(table, role, reason string) { if m == nil { return diff --git a/internal/stream/roweval.go b/internal/stream/roweval.go index 3722956a..85fd6939 100644 --- a/internal/stream/roweval.go +++ b/internal/stream/roweval.go @@ -26,13 +26,17 @@ const ( // author's to fix: a stored value the expression cannot read, or a filter // constant the column's type cannot read (a claim rendering as "abc" // against a Decimal column answers code 53 per row, against a Float one - // 72; an integer column's strict cast answers false instead). + // 72; an integer column's strict cast answers false instead). It is also + // classifyPrepare's label for a Prepare failure that is neither + // ReasonUnavailable nor ReasonDrift: the type layer refusing the parse call + // as a whole. ReasonError = typelayer.ReasonError // ReasonDecline: no verdict was reached. The expression would not compile // for this generation, chtypes would not answer for the row (one that does - // not parse lands here), or the predicate reads a column the event does not - // carry (see engineRowView.Visible). Withheld, like every answer that is not - // a definite true. + // not parse lands here, not in Prepare's error: see + // TestEngineEvaluator_UnreadableRowIsDeclined), or the predicate reads a + // column the event does not carry (see engineRowView.Visible). Withheld, + // like every answer that is not a definite true. ReasonDecline = typelayer.ReasonDecline // ReasonUnavailable: no compiled schema can answer for this tenant's table // — the tenant is not bound yet, its server line has no artifact, a compile diff --git a/internal/stream/roweval_test.go b/internal/stream/roweval_test.go index 43eaa9b4..0cdd771a 100644 --- a/internal/stream/roweval_test.go +++ b/internal/stream/roweval_test.go @@ -259,10 +259,40 @@ func TestWithheldReason_ClassifiesTypeLayerErrors(t *testing.T) { WithheldReason(classifyPrepare(&typelayer.Unavailable{Tenant: tenant.Default, Table: "clicks", Cause: "no artifact"}))) assert.Equal(t, ReasonDrift, WithheldReason(classifyPrepare(fmt.Errorf("wrapped: %w", typelayer.ErrColumnsDrift)))) - assert.Equal(t, ReasonError, WithheldReason(classifyPrepare(errors.New("unparseable row")))) + assert.Equal(t, ReasonError, WithheldReason(classifyPrepare(errors.New("the parse call was refused")))) assert.Equal(t, ReasonError, WithheldReason(errors.New("an evaluator that names no reason"))) } +// TestEngineEvaluator_UnreadableRowIsDeclined: a row the type layer cannot +// read is not a Prepare failure. Prepare succeeds, the filter does not answer +// for the row, and the withhold is labelled decline, whatever the shape: a +// value the column cannot read, a short row, a truncated array, an object, not +// JSON at all (measured on the 26.8 artifact). The hub refuses the last four +// before Prepare, as they do not pair with the column list; they are here to +// pin the label. +func TestEngineEvaluator_UnreadableRowIsDeclined(t *testing.T) { + t.Parallel() + eval := NewRowEvaluator(typelayertest.TestEngine(t, + chtypesTable("clicks", col("page", "UInt32"), col("secret", "String"), col("tenant_id", "String")))) + cols := []string{"page", "secret", "tenant_id"} + perms := policy.Evaluate(filteredPolicy(), "viewer", "clicks", "select", map[string]any{"tenant": "acme"}) + + view, err := eval.Prepare(tenant.Default, "clicks", cols, json.RawMessage(`[1,"x","acme"]`)) + require.NoError(t, err) + visible, reason := view.Visible(perms) + view.Close() + require.True(t, visible, "control: a readable row is judged: %s", reason) + + for _, row := range []string{`["abc","x","acme"]`, `[-1,"x","acme"]`, `[1,"x"]`, `[1,"x","acme"`, `{"page":1}`, `not json`} { + view, err := eval.Prepare(tenant.Default, "clicks", cols, json.RawMessage(row)) + require.NoError(t, err, row) + visible, reason := view.Visible(perms) + view.Close() + assert.False(t, visible, row) + assert.Equal(t, ReasonDecline, reason, row) + } +} + // TestEngineEvaluator_TenantsAreIndependent: the production evaluator resolves // the EVENT's tenant's table. A tenant the type layer has not bound is // unavailable on its own, while a bound tenant's row of the same table name is diff --git a/internal/typelayer/filter_test.go b/internal/typelayer/filter_test.go index ae14e19b..8ee35149 100644 --- a/internal/typelayer/filter_test.go +++ b/internal/typelayer/filter_test.go @@ -430,8 +430,8 @@ func TestVisibleWithReason_LabelsTheCause(t *testing.T) { // TestParseRow_ColumnsDriftIsAnError: a column list no INSERT into this // generation could name is drift, not a parse attempt. A row whose list is // fine but whose values do not fit it is not drift: the block holds the row's -// refusal and every predicate over it withholds (measured on the 26.6 -// artifact: ParseBlock reports no call-level error for a malformed row). +// refusal and every predicate over it withholds (measured on the 26.6 and +// 26.8 artifacts: ParseBlock reports no call-level error for a malformed row). func TestParseRow_ColumnsDriftIsAnError(t *testing.T) { eng := testEngine(t, rowsTable()) tbl, err := eng.Table(tenant.Default, "rows") From 7ad1be82574a577faa09cb7d29dc268f6cc69d54 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 12:03:23 -0400 Subject: [PATCH 59/70] docs: who labels each withheld row, and what the type layer holds in memory The stream's row view supplies error as well as filter and decline: ClickHouse raising while it evaluates the predicate over the row. Deployment's size guidance now covers the compiled parser handles and their filter caches (up to 4096 filters a handle, about 43 MiB a full handle of one-clause filters), not only the library, and the architecture page says the cache is per handle. The README's quick-start link points at the heading's real anchor. Co-Authored-By: Claude Opus 5.5 --- README.md | 2 +- docs/src/content/docs/architecture.md | 4 ++-- docs/src/content/docs/deployment.md | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 879abbda..4f1d0d6d 100644 --- a/README.md +++ b/README.md @@ -25,7 +25,7 @@

Docs · - Quick start · + Quick start · Why WaveHouse · Discussions

diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 452c44ef..f5182bbb 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -107,7 +107,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)) lives next to the keepalive primitives it shares. One abstraction per file. - **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row. Visibility itself is decided by `internal/typelayer` (`Table.ParseRow` once per event, on the event's tenant's table set, then `Row.Visible` per subscriber) — the same ClickHouse parsing and comparison semantics the server's own `WHERE` clause applies, for every column type, rather than a hand-written per-type comparator; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (one the inserting role cannot write, or a `MATERIALIZED`/`ALIAS` column; reason `decline`, though `/v1/query` still returns the row), a schema drift between the event and the live table, or an engine that is unavailable for that tenant all withhold the row rather than guessing. Each row withheld this way increments `wavehouse_sse_rows_withheld_total{table,role,reason}`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and, under a row filter, preparing one parsed row per replayed event (closed at once, since a gap-fill can run for thousands of events). `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. -- **roweval.go** — `RowEvaluator` / `RowView`, the one place stream row visibility under a role's row `filter` is decided: `Prepare` parses an event once through `internal/typelayer`, and the view answers per subscriber. A nil evaluator fails closed, and `WithheldReason` maps a prepare error to its `reason` label of `wavehouse_sse_rows_withheld_total` (`unavailable`, `drift` or `error`); the view's `Visible` supplies `filter` and `decline`. +- **roweval.go** — `RowEvaluator` / `RowView`, the one place stream row visibility under a role's row `filter` is decided: `Prepare` parses an event once through `internal/typelayer`, and the view answers per subscriber. A nil evaluator fails closed, and `WithheldReason` maps a prepare error to its `reason` label of `wavehouse_sse_rows_withheld_total` (`unavailable`, `drift` or `error`); the view's `Visible` supplies `filter`, `decline` and `error` (ClickHouse raising while evaluating the predicate over the row). - **subscriber.go** — `Subscriber`, the per-connection handle. It carries the connection's JWT claims, fixed at construction (`NewSubscriber(claims, metrics)`, no setter) — the claims the `Hub` resolves a role's row-level `filter` against, and immutability is what makes the fan-out's unsynchronized claims read race-free structurally. It owns a single ready-to-write outbound queue of `Frame`s (each tagged with its `kind`, so the handler labels the write where it happens): producers — the keepalive wheel and the event `Hub` — fan frames in with `Send` (non-blocking; a full queue drops, and `Send` itself counts the drop by frame kind, so no producer can forget to), and the handler drains `Frames()` to the client verbatim. The queue is sized for buffering live events (cap 64, up from the keepalive-only cap 1; #152 will make it a knob), and `Evict` closes its `Evicted()` channel, once, for the handler to end the stream: the `Hub`'s `Prune` does for a tenant no longer served, and the slow-consumer follow-up will for a wedged consumer. - **bucket.go** — `Bucket`, the reusable fan-out primitive: a concurrency-safe set of subscribers. `Push` fans one `Frame` to every member fire-and-forget — the keepalive wheel's ring is its only caller now that both `Hub` paths iterate `Snapshot`, since the schema announcement is per connection even where the projection is shared per role; `Snapshot` exposes the members so the event `Hub` can evaluate row visibility per subscriber before sending (drop counting lives in `Send` itself). The `Hub` holds one `Bucket` per `(topic, role)` so a projected frame is built once and sent to every member instead of re-projected per subscriber. - **heartbeat.go** — The keepalive wheel (`Heartbeater`). A single process-wide ticker fans a minimal `:` comment across the ring of `Bucket`s, waking ~1/N of live streams per tick so the writes don't synchronize. The effective per-connection keepalive period is `stream.keepalive_interval` in the settings directory (the wheel ticks every `keepalive_interval ÷ keepalive_buckets`, so one rotation spans the interval; a reload calls `Reconfigure`, which rebuilds the ring in place with every live subscriber carried over); the owning handler goroutine does the actual write, so the shared ticker never touches a `ResponseWriter` directly. @@ -172,7 +172,7 @@ Each tenant has its own table set inside that engine, bound from the tenant's ow - **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind. A tenant whose ClickHouse line has no installed artifact is unavailable on its own, and so is one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Either way the cause is recorded (an ingest client sees only a generic `503`; the cause goes to the log), every other tenant keeps working, and a later bind of the same tenant (a reload, a refresh that now agrees) clears it. - **`Engine.RoleTable(tenant, table, shape)`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column, including one the role may not otherwise write — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column stays in the schema as `MATERIALIZED` of its default, so naming it is ClickHouse's code 117 while expressions that read it still compile, and an absent check column takes the claim as its default while a supplied value is accepted only if it equals the claim. A role-shape pool starts at one handle and grows to `min(GOMAXPROCS, 4)` when every handle is busy; a base table's grows to `min(GOMAXPROCS, 8)`. - **`Table.IngestWith(format, opts, body, checks...)`** (`Ingest` is the same with no options) runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. -- **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per table. Only a definite true admits; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (`decline`), schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. +- **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per handle (up to 4096 on each). Only a definite true admits; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (`decline`), schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. - A tenant that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row of that tenant's tables from a role that has a row `filter`, with reason `unavailable`. - A role whose schema cannot be compiled — or that may write no column of the table — is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`, and the cause is logged once a minute; it is not the `503` an unavailable tenant gets, since retrying cannot fix a policy or a table. A shape that fails only because an injected check value cannot be its column's default (`'abc'` on a `UInt64`) is retried without the defaults and served: a record omitting that column then fails the check (`403` on an integer column, `422` on another), and only a shape that does not compile even without them gets the `500`. - **`typelayertest`** (`internal/typelayer/typelayertest`) holds the test helpers other packages' tests use — `TestEngine`, `SkipWithoutArtifact`, and `RequireEnv` (the `WAVEHOUSE_TEST_REQUIRE_CHTYPES` switch that turns a skip into a failure) — which only test binaries link (`typelayer` itself never imports `testing`). The `typelayer` package's own tests use a small in-package copy, since they cannot import a package that imports them. diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index e07e2d68..722379a2 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -120,7 +120,7 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ **Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse never fetches an artifact itself: a line with no installed artifact makes the tenants on it unavailable at their schema refresh, and boot is refused only when no artifact is installed at all, or when an explicit `clickhouse.chtypes_registry` does not exist or cannot be read. One exception: the SDK honors `CHTYPES_AUTOFETCH=1` from the environment over WaveHouse's setting, and would then download a missing line (hundreds of MB) inside a schema refresh — leave it unset. -**Size.** Each artifact is roughly 160–300 MB on disk; a running process holding several loaded versions (e.g. across a rolling ClickHouse upgrade) costs roughly 120 MB of resident memory per loaded version (the chtypes multi-version guide's figure; a library is opened on first use of its line, not at registry construction). +**Size.** Each artifact is roughly 160–300 MB on disk; a running process holding several loaded versions (e.g. across a rolling ClickHouse upgrade) costs roughly 120 MB of resident memory per loaded version (the chtypes multi-version guide's figure; a library is opened on first use of its line, not at registry construction). On top of the library, every table of every tenant an API process serves holds compiled parser handles, roughly 40 KiB each and 96 KiB once warm: one from the moment its schema is bound, more only while every handle is busy, up to `GOMAXPROCS` and at most 8, plus up to 256 role shapes per table (one per distinct set of writable columns and claim-stamped defaults among the roles that insert) of up to 4 handles each. Each handle also caches up to 4,096 compiled filters, one per distinct expression and claim values, row filters and insert checks alike: about 10 KiB for a one-clause filter and 27 KiB for two clauses with a three-value `in` (measured on the 26.8 artifact, darwin arm64), so a full cache of one-clause filters is about 43 MiB, and a table's 8 base handles about 340 MiB. A cache fills only once that many distinct claim sets have been evaluated on one handle — a row-filtered table streamed to thousands of subscribers whose claims differ — and then stays full, evicting the least recently used. All of it multiplies by tables and by tenants. **Docker images** ship the artifact(s) baked in: the image build fetches the lines `scripts/fetch-chtypes.sh` lists (`LOCK_LINES`), each pinned by `chtypes.lock` (see below), so a container never needs network access to chtypes' artifact store at runtime. The image sets `CHTYPES_REGISTRY=/opt/chtypes/artifacts` (the SDK's own variable); set `WH_CHTYPES_REGISTRY` only to point at a bind-mounted directory instead. **The published images support ClickHouse 26.8 only**: they bake only that line, and a server on any other line answers every tenant on it `503` (with row-filtered streams withholding their rows) behind a generic body. For another line, fetch it for the container's platform into a directory of its own — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --platform linux-amd64 --dest ./chtypes-artifacts ` (`linux-arm64` on arm) — make it readable by the image's user (`chmod -R a+rX ./chtypes-artifacts`; the fetch writes each line's directory with mode `0700`), mount it read-only, and set `WH_CHTYPES_REGISTRY` to the mount path. That directory is searched first and the baked line stays available; a path that does not exist or cannot be read refuses boot. From 871c86c2f48bc47e88c78623449e669c04807ac6 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 12:08:02 -0400 Subject: [PATCH 60/70] docs: the envelope re-encodes the writer's row without changing any value json.Marshal compacts the row and escapes <, > and &, so "the exact bytes" overstated it; every decoded value is still ClickHouse's. Co-Authored-By: Claude Opus 5.5 --- docs/src/content/docs/api.md | 2 +- internal/ingest/types.go | 8 +++++--- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 2d2a9d9e..8511e0a8 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -915,7 +915,7 @@ The request that produced this envelope omitted `received_timestamp` (`DEFAULT n | `received_timestamp` | string | RFC 3339 nano timestamp when WaveHouse received the event. | | `format` | string | Row format. Always `JSONCompactEachRow` today; stated on the wire so a reader can tell an envelope it understands from one it doesn't. | | `columns` | string[] | The table's **wire** column names, in declaration order — what each position in `row` means (`internal/typelayer.Table.WireColumns`: the table's columns minus any `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column and minus any the role may not write — none of the three kinds is ever part of a published row). | -| `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order — the exact bytes ClickHouse's own writer produced for this stored row (`internal/typelayer`'s `Table.IngestWith`, via chtypes). A column the request body omitted carries its evaluated `DEFAULT`, or the type's default where none is declared (`null` only on a `Nullable` column) — the same as a native `INSERT` naming fewer columns than the table has. `DateTime`/`DateTime64` values are ClickHouse's own rendering (see [Timestamp rendering](#timestamp-rendering)), and an out-of-range integer is wrapped the way a real `INSERT` wraps it. | +| `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order — ClickHouse's own writer output for this stored row (`internal/typelayer`'s `Table.IngestWith`, via chtypes); the envelope's JSON encoding only drops insignificant whitespace and escapes `<`, `>` and `&` as `\u003c`, `\u003e`, `\u0026`, so every value decodes to exactly what ClickHouse wrote. A column the request body omitted carries its evaluated `DEFAULT`, or the type's default where none is declared (`null` only on a `Nullable` column) — the same as a native `INSERT` naming fewer columns than the table has. `DateTime`/`DateTime64` values are ClickHouse's own rendering (see [Timestamp rendering](#timestamp-rendering)), and an out-of-range integer is wrapped the way a real `INSERT` wraps it. | `columns` and `row` are only meaningful together: a reader that cannot pair them — a length mismatch, an undecodable row, a `columns` list naming one column twice — has no way to map a value to a column. Both readers also refuse an envelope whose `format` they do not recognize. Either way the SSE fan-out withholds such an envelope rather than guess, and the batch consumer parks it on the DLQ with `X-DLQ-*` headers — acking and dropping it only where the DLQ is switched off for that table, since it can never insert on retry. Both outcomes increment `wavehouse_ingest_poison_total`, separated by its `disposition` label (`parked` / `dropped`). diff --git a/internal/ingest/types.go b/internal/ingest/types.go index 7a8fd2a0..1bb365e0 100644 --- a/internal/ingest/types.go +++ b/internal/ingest/types.go @@ -15,9 +15,11 @@ const FormatJSONCompactEachRow = "JSONCompactEachRow" // EventMessage is the wire format published to the MQ. // // Row data travels POSITIONALLY: Row is one JSONCompactEachRow line (a JSON -// array, no trailing newline) as ClickHouse's own writer produced it, and -// Columns names its positions — the table's wire columns, or the narrower list -// a column-restricted role writes, in declaration order. The two are only +// array, no trailing newline) as ClickHouse's own writer produced it — the +// envelope's json.Marshal only compacts it and escapes <, > and &, which +// changes no decoded value — and Columns names its positions: the table's wire +// columns, or the narrower list a column-restricted role writes, in +// declaration order. The two are only // meaningful together — a reader that cannot pair them (a length mismatch, an // undecodable row, a repeated column name) has no way to map a value to a // column and must fail closed rather than guess. From 241f869edff03593ffd67e69abe6460e6d4cee32 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 12:17:08 -0400 Subject: [PATCH 61/70] docs: finish the exact-bytes correction; say what the error reason means Two more copies of "the exact bytes" described the envelope's row, which json.Marshal compacts and escapes; the stream's error reason is a parse call refused as a whole, not a row it cannot read. The CHANGELOG's file list for the slash pin names only the files the PR changes. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- CHANGELOG.md | 2 +- docs/src/content/docs/access-control.mdx | 2 +- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/ingest-pipeline.md | 2 +- internal/ingest/types.go | 8 ++++---- 6 files changed, 9 insertions(+), 9 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 5c1324e6..57bb177d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -38,7 +38,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`coord/`** — leases for work that must run in one process at a time (`Observer.Held` reads whether one is held without campaigning): `Coordinator.TryAcquire(ctx, name)` → a `Term` (fencing `Token`, strictly increasing per name; `Done`/`Err`, `ErrLost` on loss; `Resign`), `ErrHeld` while another holder's — or this coordinator's own — term is live; `RunElected` runs a loop only while holding its lease, resigning when the loop returns and campaigning again every `RetryPeriod`. `Local` is the in-process implementation (first taker wins, never expires; `Peer` is a second handle over the same table for tests); every implementation runs `coordtest.Conformance`. Imports only the standard library, so a distributed backend lives beside its connection: `coord.backend: nats` is `internal/mq/lease.go` (`ExternalNATS.Leases`), a key per lease in the operator's KV bucket, the KV revision as the fencing token, and expiry judged on the candidate's own clock (the same revision seen unchanged for 15s), never by a server TTL. `internal/app`'s `wireCoord` opens the one `coord.backend` selects and the sweeper runs through `RunElected` under the `sweeper` lease - **`dedupe/`** — `Deduplicator` interface (two-phase `Reserve`/`Commit`/`Release` over `Key{Table, ID}`; every backend passes the `dedupetest` conformance suite) → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant and table, pending claims in memory, committed ids stored with their expiry and deleted by an hourly background sweep along with the version-0 keys from before the table joined the key, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges) or `Dynamo` (one shared DynamoDB table, conditional `PutItem` claims; conformance-tested against dynamodb-local, selected by `dedupe.backend: dynamodb`; boot checks the table and never creates it outside dynamodb-local), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` or, gated on the table check (`Factory.Gated`), `Dynamo.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) - **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`, and starts a tenant over on a fresh registry when a reload moves it to another address or database: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version and default timezone, and fires an `OnRefresh` hook after every refresh it publishes and before the registry reports itself loaded (overlapping refreshes publish in the order they started: one that finishes after a later-started one has published is dropped, so an older snapshot never replaces a newer one), so a loaded tenant is a bound one — `typelayer.Engine.Bind` is its only consumer (Key Design Decision #21) -- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else; `row` is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced for that stored record (via `typelayer`'s `Table.IngestWith`), `columns` names its positions (the table's wire columns, `typelayer.Table.WireColumns`: no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` column — or the narrower list a column-restricted role produces); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs with `typelayer.InsertSettings()` plus `async_insert=0` (it never loads the artifact, so an ingest-only process needs none). In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) +- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output; over a `Sharded` queue, `claims.go`'s `ClaimShards` narrows the worker to the units this process is assigned — membership leases, capped rendezvous, halt-drain-then-release handover and stop, reset at takeover from a dead owner, each unit's share of a 10,000-row budget of unsettled rows). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else; `row` is ClickHouse's own `JSONCompactEachRow` writer output for that stored record (via `typelayer`'s `Table.IngestWith`; the envelope's `json.Marshal` only compacts it and escapes `<`, `>`, `&`), `columns` names its positions (the table's wire columns, `typelayer.Table.WireColumns`: no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` column — or the narrower list a column-restricted role produces); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs with `typelayer.InsertSettings()` plus `async_insert=0` (it never loads the artifact, so an ingest-only process needs none). In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`, so non-admin callers never reach the proxy) or through an operator-authored pipe that writes, gated only by its `allowed_roles` (#386). A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) - **`keyenc/`** — the one escaping composite keys are built from: `Escape` keeps `[A-Za-z0-9_-]` (exactly the tenant-id grammar, so a tenant id is its own escaped form) and writes every other byte as `%XX`, `Unescape` is `url.PathUnescape` (lenient: either hex case, and a byte left unescaped reads as itself, so a `%2D` an earlier build wrote still reads), `Join`/`AppendJoin` escape each field and put a separator between them (they panic on no fields, and on a separator the escaping could write or one outside ASCII) and `Split` reverses them. The package that builds a key takes raw names and escapes them itself, so no caller has to and no field reaches a key unescaped: NATS subjects (`Join`/`Split` after the verbatim tenant) and the cache's keys — the version index and the shared backend's Redis keys (`internal/cache`) — and the dedupe keys (`/
/`) use it; changing what it keeps orphans every stored key (an orphaned dedupe key lets a seen id through again), and on the shared backend, whose keys every process builds for itself, splits them between builds for the length of a rolling upgrade (a bump one build makes misses the entries the other filed, served until their TTL) - **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20), and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal, `ErrUnavailable` a broker that cannot be reached — both a `503`, with `Retry-After` `30` and `5`; `WithIdempotencyKey` makes a republish inside the queue's duplicate window a no-op), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). A broker whose ingest queue is split into units one consumer at a time owns implements `Sharded` too (`IngestUnits`, `ResetOrphaned`, `Unowned`; `ConsumerConfig.Units` narrows a consumer to some of them, and its consumer is a `Releaser` and a `Halter`, capped per unit by `ConsumerConfig.MaxHeld`, with `ErrConsumerMismatch` when an operator durable no longer fits), and `Message.OnSettled` runs a hook once, at the first ack or nak attempt, confirmed or not. Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, stream names, sequences, and ack floors are private to the implementations, whose subject tokens are escaped by the shared `internal/keyenc`: `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`, `deadletter.go`), which `internal/app` constructs and hands everything else as a `mq.Broker`, and `ExternalNATS` (`external.go`, `subject_nats.go`, `nats_topology.go`: an operator-owned cluster whose streams, durables and lease bucket it never creates, changes, purges or deletes; `lease.go` holds `coord.backend: nats`'s leases in that bucket), which `internal/app` constructs from the `mq.nats` block when `mq.backend` is `nats` ([#613](https://github.com/Wave-RF/WaveHouse/issues/613)). Every implementation passes the conformance suite in `internal/mq/mqtest` (`mqtest.Run`), which states the `Broker` contract as behavior; a new backend runs it from its own test, with `mqtest.Caps` only where its semantics legitimately differ - **`observability/`** — OpenTelemetry pipeline: `InitProvider` wires trace/metric/log providers via OTLP gRPC (each signal independently gated). A top-level `Prometheus` config block drives an optional `/metrics` scrape endpoint that runs independently of OTLP push — standalone (Alloy/Mimir scrape, no collector), alongside OTLP, or off. `NewLogger` produces a slog handler that fans out to stdout AND OTLP (stdout always 100%, OTLP sample-rate-aware). `TraceHandler` injects trace_id/span_id from active spans. `tracer.go` provides W3C trace context propagation over message headers (`InjectHeaders`/`ExtractHeaders` on a plain header map; `internal/mq` injects on every publish and extracts on the `Subscribe` path, so this package never sees a NATS type). diff --git a/CHANGELOG.md b/CHANGELOG.md index 041088b2..7754ea00 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -100,7 +100,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Structured queries and pipes are rendered by ClickHouse, not by WaveHouse** (BREAKING; `internal/api/sql_classify.go` (was `clickhouse_exec.go`, trimmed to the mutation classifier), `internal/api/clickhouse_http.go` (new), `internal/api/{structured_query,pipes,cache_key,ch_settings}.go`, `internal/query/builder.go`, `internal/chsql/chsql.go`, `internal/settings/validate.go`): `POST /v1/query` and `GET/POST /v1/pipes/{name}` used to run through `clickhouse-go`'s native driver and re-render every row in Go; they now go over the tenant's ClickHouse HTTP interface with `default_format=JSONEachRow`, bind each value as a named `{pN:String}` parameter, and the cache stores ClickHouse's own bytes. **`Decimal*` values are now a JSON number (`12.5`) where they were a string (`"12.5"`)**, `DateTime` is spelled by ClickHouse as RFC 3339 in UTC (`"2026-06-21T04:00:00.123Z"`, the fraction at the column's precision), `Date`/`Date32` are `"2026-06-21"` where the native driver gave `"2026-06-21T00:00:00Z"`, and `NaN`/`Inf` are `null` where they were a `500`. Response object keys come back in **SELECT order** rather than alphabetical. Every read runs with `readonly=2` (write pipes do not), a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `wait_end_of_query=1` and pinned rendering settings, so a statement the mutation classifier missed cannot write through a read path and a runaway query is stopped by ClickHouse. A `null` filter value is now `400 {"error":"filter value must not be null"}` instead of a silently empty result (`col = NULL` is never true), and an `in` list travels as a ClickHouse external table, so its size is bounded only by the 1 MiB request body. A filter value on a `Date`/`DateTime` column is parsed by ClickHouse (see Fixed). Failure classification keeps the same `code`/`retryable` table of the query paths, with one addition: a response past 64 MiB is now `502 clickhouse.response_too_large` on `/v1/query` and pipes, where the native path had no cap and a large result simply came back. The reader's HTTP connections are capped per connection tuple (URL, user, password, database, TLS) at the largest `max_open_conns` among the tenants sharing it, and its requests carry the tenant's `clickhouse.headers`; the settings warning about a plaintext HTTP hop (`internal/settings/validate.go`) now says it carries the credentials on every query and insert. The 64 MiB cap is new on these two paths, so a pipe (which has no row limit) returning more than 64 MiB now fails; `/v1/query` normally stays under it through its default row cap. Cache keys change value, so a deploy serves one cold cache and an old and a new build never share a Redis entry; `X-Cache` semantics, the namespace deps and the singleflight are untouched, and a pipe that writes still bypasses the cache. `/v1/ops/query` is unaffected. -- **ClickHouse-rendered JSON leaves `/` unescaped on every path** (`internal/api/{clickhouse_http,query,ingest_framing}.go` (+ tests), `internal/typelayer/ingest.go` (+ tests), `tests/integration/query_types_test.go` (+ its pin), `docs/src/content/docs/{api,architecture,getting-started}.md`, `docs/src/content/docs/pipes.mdx`, `AGENTS.md`): ClickHouse's JSON writer spells `/` as `\/` by default, and a tenant's profile could choose either. `output_format_json_escape_forward_slashes=0` is now pinned on `/v1/query`, pipes, `/v1/ops/query` and the ingest export behind the NATS/SSE row, so `/v1/query`, pipes and the stream keep the `"/home"` spelling of earlier releases, and `/v1/ops/query` changes from ClickHouse's default `"\/home"` to `"/home"` (a `SETTINGS` clause in the SQL sent to `/v1/ops/query` still overrides it). The cache's rendering marker moved to `JSONEachRow/3`, so a deploy starts with a cold query cache. +- **ClickHouse-rendered JSON leaves `/` unescaped on every path** (`internal/api/{clickhouse_http,query,ingest_framing}.go` (+ tests), `internal/typelayer/ingest.go` (+ tests), `tests/integration/query_types_test.go` (+ its pin), `docs/src/content/docs/{api,architecture}.md`, `AGENTS.md`): ClickHouse's JSON writer spells `/` as `\/` by default, and a tenant's profile could choose either. `output_format_json_escape_forward_slashes=0` is now pinned on `/v1/query`, pipes, `/v1/ops/query` and the ingest export behind the NATS/SSE row, so `/v1/query`, pipes and the stream keep the `"/home"` spelling of earlier releases, and `/v1/ops/query` changes from ClickHouse's default `"\/home"` to `"/home"` (a `SETTINGS` clause in the SQL sent to `/v1/ops/query` still overrides it). The cache's rendering marker moved to `JSONEachRow/3`, so a deploy starts with a cold query cache. - **WH001 (no hard-wrapped prose) now applies to every tracked Markdown file, with no carve-out** (`.github/.markdownlint.json` (deleted), `.claude/.markdownlint.json` (deleted), `.claude/skills/integration-astro-view-transitions/` (deleted), `.markdownlint-cli2.jsonc`, `.github/workflows/README.md`, `.claude/skills/pm-triage/references/routine.md`, `AGENTS.md`, `scripts/docs-prose.sh`, `.github/prompts/docs-review.md`, `docs/src/content/docs/claude-code.md`, `docs/src/content/docs/development.md`, `.claude/agents/docs-reviewer.md`): two path-scoped configs had switched WH001 off under `.github/` and `.claude/` ever since [#489](https://github.com/Wave-RF/WaveHouse/pull/489) introduced the rule — baked in from the start rather than added in response to a discovered problem — which left the repo documenting the rule three ways and disagreeing with itself: `CONTRIBUTING.md` promises contributors `make lint` enforces it *everywhere*, while `AGENTS.md` and the `.markdownlint-cli2.jsonc` header wrote up the carve-out. Not theoretical: on [#520](https://github.com/Wave-RF/WaveHouse/pull/520) a reviewer correctly flagged a hard-wrapped bullet in `.github/workflows/README.md`, an agent pointed at `"WH001": false` for that path and pushed back, and the reviewer recorded a *learning* never to flag WH001 there — the wrong invariant, learned off the wrong side of the contradiction ([#521](https://github.com/Wave-RF/WaveHouse/issues/521)). Both configs are deleted — each held nothing but the override, so the root `.markdownlint.json` governs again — and the 51 hard-wrapped paragraphs they were hiding are joined: 41 in `.github/workflows/README.md` and 10 in `.claude/skills/pm-triage/references/routine.md`, mechanical joins with no wording changed and every fenced block, table row, and heading byte-identical either side of the reflow. Deleted with them: the wizard-installed PostHog skill at `.claude/skills/integration-astro-view-transitions/` — 9 files, ~1,456 lines, including an 809-line `EXAMPLE.md` copied wholesale from `PostHog/context-mill`. Its integration job finished in [#277](https://github.com/Wave-RF/WaveHouse/pull/277), nothing in the repo calls it, and the docs-site setup it once described is documented where it belongs — in `docs/src/components/PostHog.astro` and this file. Keeping unowned third-party prose in the tree means content that drifts silently on every upstream bump and that nobody here reviews; it was also the single file that would have needed a special-case lint exclusion, so removing it is what lets WH001 apply with **no exception at all** rather than one documented one. Its two inventory rows in `claude-code.md` go with it, as does the now-dead `docs/posthog-setup-report.md` entry in the `scripts/docs-prose.sh` denylist (the wizard's other artifact, deleted back in [#502](https://github.com/Wave-RF/WaveHouse/pull/502)) and the copies of that denylist in `AGENTS.md` and `.github/prompts/docs-review.md`, which the script's header requires be kept in lockstep. Review of the change then turned up four more things the exclusion had been hiding, all fixed here: **WH001 has a blind spot** — `no-hard-wrapped-prose.mjs` classifies any line indented four or more spaces as an indented code block, so a *nested* list item is never joined, which left three hard-wrapped bullets in `.github/workflows/README.md` §"Adding a job" that the autofix could not see (unwrapped by hand; they were the last hard-wrapped prose paragraphs in the repo) and made `AGENTS.md`'s and `development.md`'s "a list item is joined as a unit" wrong for nested items (both now state the four-space caveat); the `scripts/docs-prose.sh` header told readers to keep its denylist in lockstep with **two** sibling copies when there are **three** — the missed one being `.claude/agents/docs-reviewer.md`, the gating subagent's own system prompt, which had in fact been silently out of sync for the whole life of the `posthog-setup-report.md` exclusion; the `.markdownlint-cli2.jsonc` header's "applies to every tracked Markdown file" was exact for WH001 but not WH002, which returns early on anything that isn't `.mdx`; and the job-graph diagram omitted `docs-deploy`'s `needs` edges from `unit`, `integration`, and `e2e`, contradicting invariant 2 three lines below it. The denylist also drops its `PERF-CLAIMS-REVIEW.md` entry — unlike the wizard artifact this one names a file that was **never tracked** at all, so it guarded a hypothetical; the list's other general cases are patterns (`*.draft.md`, `*.old.md`) that already cover a one-off review document, and a literal filename restated in four places is the outlier. `scripts/docs-prose.sh all` still resolves the same 27-file prose set. diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 8da838d8..197ed1e1 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -367,7 +367,7 @@ The same policy drives every data path, but not every field is meaningful on eve :::caution[Live streams enforce column and row policy, but not resource limits] SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause on a server running that line's default settings, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72), or the parse of the event failed outright; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72), or the parser refused the call as a whole, which is not a property of the row; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index f5182bbb..4ea1c5e5 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -181,7 +181,7 @@ See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest e ### `ingest/` — Ingest Pipeline, DLQ & Sweeping -- **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line — the exact bytes ClickHouse's own writer produced for the stored record — with `columns` naming its positions (the table's wire columns, `typelayer.Table.WireColumns`: no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` column — or the narrower list a column-restricted role produced); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with `typelayer.InsertSettings()` (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) plus `async_insert=0`, the same parsing settings chtypes compiled the row with. The worker never loads the artifact: `InsertSettings` is static. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so under `mq.backend: embedded` the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Under `mq.backend: nats`, anyone the operator lets publish to `.ingest.>` reaches it too, past auth, policy and schema validation, so that right belongs to the `wavehouse` user alone. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy — or through an operator-authored [pipe that writes](/pipes#pipes-that-write), gated only by its `allowed_roles`. A batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling) is never tried: no row of it could pass, so `parkBatch` takes it to the DLQ switch whole, logging once per batch rather than twice per row. Otherwise a bulk-insert failure is first classed by `chconn.Classify`: a ClickHouse that cannot take the insert (unavailable, denied, or no verdict at all) sends the batch back to the MQ for a delayed redelivery (`retryLater` → `mq.Message.NakWithDelay`), under a backoff shared by every table on the same pool (a failure of one table — read-only, too many parts — backs off that table alone), and never to the DLQ — the same when it stops answering mid-isolation. Only when ClickHouse rejects the batch, or refuses a multi-row batch for its size (`chconn.Splittable`: too many partitions for one INSERT, the memory limit), is it re-inserted row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. +- **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line — ClickHouse's own writer output for the stored record, re-encoded by the envelope's JSON encoding without changing any value — with `columns` naming its positions (the table's wire columns, `typelayer.Table.WireColumns`: no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` column — or the narrower list a column-restricted role produced); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with `typelayer.InsertSettings()` (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) plus `async_insert=0`, the same parsing settings chtypes compiled the row with. The worker never loads the artifact: `InsertSettings` is static. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so under `mq.backend: embedded` the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Under `mq.backend: nats`, anyone the operator lets publish to `.ingest.>` reaches it too, past auth, policy and schema validation, so that right belongs to the `wavehouse` user alone. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy — or through an operator-authored [pipe that writes](/pipes#pipes-that-write), gated only by its `allowed_roles`. A batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling) is never tried: no row of it could pass, so `parkBatch` takes it to the DLQ switch whole, logging once per batch rather than twice per row. Otherwise a bulk-insert failure is first classed by `chconn.Classify`: a ClickHouse that cannot take the insert (unavailable, denied, or no verdict at all) sends the batch back to the MQ for a delayed redelivery (`retryLater` → `mq.Message.NakWithDelay`), under a backoff shared by every table on the same pool (a failure of one table — read-only, too many parts — backs off that table alone), and never to the DLQ — the same when it stops answering mid-isolation. Only when ClickHouse rejects the batch, or refuses a multi-row batch for its size (`chconn.Splittable`: too many partitions for one INSERT, the memory limit), is it re-inserted row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. - **backoff.go** — The retry backoff behind `retryLater`: a small circuit breaker per ClickHouse pool (the target's URL, user and database), and one per pool and table for a failure of one table (`chconn.TableScoped`). A failure opens it for 1 s, doubling to a 30 s cap, each window jittered down to half; while it is open, flushes and arriving rows are handed back without a request, and once it elapses one flush probes. Any answer that is not an outage closes it. - **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. `Row` is the exact bytes `internal/typelayer`'s `IngestWith` returned for an accepted record — ClickHouse's own `JSONCompactEachRow` writer output, `DEFAULT`s already filled in — not a value WaveHouse encodes itself. - **claims.go**, **assign.go** — `ClaimShards` wraps a `Sharded` queue so an ingest process consumes only its share of the units, one process per unit at a time. Each process holds one membership lease, `ingest.m` for the lowest free `j` under the number of configured units (at most 64, read 16 at a time each tick), and every tick (2s) reads which slots are live (`coord.Observer.Held`); `assignUnits` gives every unit to a live slot by rendezvous hashing capped at ⌈units/slots⌉, the same result in every process for the same view; the extras are assigned apart from the configured units, so processes whose lists of extras differ (each refreshes it every five minutes) still agree on every configured unit. A unit no longer assigned halts (`mq.Halter`: stops fetching, keeps its pin, returns once what it fetched reached the worker), waits (bounded by the worker's 60s ack wait, the shortest `ack_wait` a durable may have) for the rows it delivered to settle (`Message.OnSettled`), then releases its pin (`mq.Releaser`). `Halt` on the claiming consumer ends the ticks and halts every unit at once; the worker calls it before its final flush, and the claims' stop releases the units and resigns the membership lease after it. A unit taken over from a slot that is no longer live is reset (`ResetOrphaned`) before it is bound — waiting, unbound, while the broker answers `ErrUnitHeld` (the dead owner's pin has not lapsed, or the unit was active within its pinned TTL), up to 30s — so the dead owner's unacked rows come back at once; on a process's first tick only if it is the one live member (a restart after a clean stop; after a crash the dead run's lease still counts as live for a lease duration). Each unit's rows delivered and unsettled are capped at its share of the worker's 10,000 (`ClaimConfig.MaxHeld`): an even share over the units assigned (`unitShare`, recomputed each tick and read by the broker before every fetch through `ConsumerConfig.MaxHeld`), never under 1,000, two batches; at its share a unit fetches only what keeps its pin, and one stuck unit never takes another's. A row stops counting at its first ack or nak attempt (`Message.OnSettled` fires whether or not the broker confirms), or after `ack_wait` if the worker never settles it, since the broker redelivers it then as a new row. A configured unit whose delivery ends, or whose durable or stream is gone or whose durable no longer fits (`mq.ErrConsumerNotFound`, `mq.ErrConsumerMismatch`) when it is bound, fails the worker; any other bind failure is retried next tick, logged as an error once it has lasted a minute, and an extra's end is logged. The lowest live slot counts the units with rows and no owner (`Sharded.Unowned`) for `wavehouse_ingest_shards_unowned`. diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index b9493940..d78a3407 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -20,7 +20,7 @@ It is deliberately detailed: this is a hot, concurrency-heavy path, and the goro | `assign.go` | `assignUnits` — rendezvous hashing capped at ⌈units/slots⌉, the same owners in every process for the same view | | `types.go` | `EventMessage` wire format and the `BufferConsumerName` constant | -The pipeline is **insert-only**. (Upgrading across the v2 envelope? [Drain the queue first](/deployment#upgrading-across-the-v2-ingest-envelope).) Each NATS message is one envelope — `{table_name, scope, received_timestamp, format, columns, row}`, documented field by field in [API → Internal Wire Format](/api#internal-wire-format-nats). What matters here: `row` is not something WaveHouse encodes, it is the exact `JSONCompactEachRow` bytes ClickHouse's own writer produced at ingest time, through the chtypes engine, and `columns` names its positions. The worker parses the envelope, groups a batch by column list, and bulk-`INSERT`s each group as `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with the settings chtypes compiled the row with plus `async_insert=0`. A role that may not write every column produces a shorter `columns` list, which is simply another batch group. Non-insert mutations go through `POST /v1/ops/query` (admin-only) or an operator-authored [pipe that writes](/pipes#pipes-that-write). +The pipeline is **insert-only**. (Upgrading across the v2 envelope? [Drain the queue first](/deployment#upgrading-across-the-v2-ingest-envelope).) Each NATS message is one envelope — `{table_name, scope, received_timestamp, format, columns, row}`, documented field by field in [API → Internal Wire Format](/api#internal-wire-format-nats). What matters here: `row` is not something WaveHouse encodes, it is ClickHouse's own `JSONCompactEachRow` writer output from ingest time, through the chtypes engine (the envelope's JSON encoding only drops insignificant whitespace and escapes `<`, `>` and `&`, so every value decodes to exactly what ClickHouse wrote), and `columns` names its positions. The worker parses the envelope, groups a batch by column list, and bulk-`INSERT`s each group as `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with the settings chtypes compiled the row with plus `async_insert=0`. A role that may not write every column produces a shorter `columns` list, which is simply another batch group. Non-insert mutations go through `POST /v1/ops/query` (admin-only) or an operator-authored [pipe that writes](/pipes#pipes-that-write). ## High-level shape diff --git a/internal/ingest/types.go b/internal/ingest/types.go index 1bb365e0..98eeca27 100644 --- a/internal/ingest/types.go +++ b/internal/ingest/types.go @@ -19,10 +19,10 @@ const FormatJSONCompactEachRow = "JSONCompactEachRow" // envelope's json.Marshal only compacts it and escapes <, > and &, which // changes no decoded value — and Columns names its positions: the table's wire // columns, or the narrower list a column-restricted role writes, in -// declaration order. The two are only -// meaningful together — a reader that cannot pair them (a length mismatch, an -// undecodable row, a repeated column name) has no way to map a value to a -// column and must fail closed rather than guess. +// declaration order. The two are only meaningful together — a reader that +// cannot pair them (a length mismatch, an undecodable row, a repeated column +// name) has no way to map a value to a column and must fail closed rather +// than guess. type EventMessage struct { TableName string `json:"table_name"` Scope string `json:"scope"` From da0c0aa266b10a66a32d7e8d62973f14d047e2a0 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 12:26:24 -0400 Subject: [PATCH 62/70] docs: the last two copies of the exact-bytes claim; untangle the error reason Co-Authored-By: Claude Opus 5.5 --- docs/src/content/docs/access-control.mdx | 2 +- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/ingest-pipeline.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 197ed1e1..4679b642 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -367,7 +367,7 @@ The same policy drives every data path, but not every field is meaningful on eve :::caution[Live streams enforce column and row policy, but not resource limits] SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause on a server running that line's default settings, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72), or the parser refused the call as a whole, which is not a property of the row; on an integer column such a claim is a `filter`), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72); on an integer column such a claim is a `filter` — and also when the event's parse (`Table.ParseRow`) is refused as a whole, which says nothing about the row), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 4ea1c5e5..edeefc15 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -183,7 +183,7 @@ See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest e - **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line — ClickHouse's own writer output for the stored record, re-encoded by the envelope's JSON encoding without changing any value — with `columns` naming its positions (the table's wire columns, `typelayer.Table.WireColumns`: no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` column — or the narrower list a column-restricted role produced); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow` with `typelayer.InsertSettings()` (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) plus `async_insert=0`, the same parsing settings chtypes compiled the row with. The worker never loads the artifact: `InsertSettings` is static. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so under `mq.backend: embedded` the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Under `mq.backend: nats`, anyone the operator lets publish to `.ingest.>` reaches it too, past auth, policy and schema validation, so that right belongs to the `wavehouse` user alone. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy — or through an operator-authored [pipe that writes](/pipes#pipes-that-write), gated only by its `allowed_roles`. A batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling) is never tried: no row of it could pass, so `parkBatch` takes it to the DLQ switch whole, logging once per batch rather than twice per row. Otherwise a bulk-insert failure is first classed by `chconn.Classify`: a ClickHouse that cannot take the insert (unavailable, denied, or no verdict at all) sends the batch back to the MQ for a delayed redelivery (`retryLater` → `mq.Message.NakWithDelay`), under a backoff shared by every table on the same pool (a failure of one table — read-only, too many parts — backs off that table alone), and never to the DLQ — the same when it stops answering mid-isolation. Only when ClickHouse rejects the batch, or refuses a multi-row batch for its size (`chconn.Splittable`: too many partitions for one INSERT, the memory limit), is it re-inserted row by row: rows that succeed are acked, and only the rows ClickHouse rejects again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. - **backoff.go** — The retry backoff behind `retryLater`: a small circuit breaker per ClickHouse pool (the target's URL, user and database), and one per pool and table for a failure of one table (`chconn.TableScoped`). A failure opens it for 1 s, doubling to a 30 s cap, each window jittered down to half; while it is open, flushes and arriving rows are handed back without a request, and once it elapses one flush probes. Any answer that is not an outage closes it. -- **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. `Row` is the exact bytes `internal/typelayer`'s `IngestWith` returned for an accepted record — ClickHouse's own `JSONCompactEachRow` writer output, `DEFAULT`s already filled in — not a value WaveHouse encodes itself. +- **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. `Row` is ClickHouse's own `JSONCompactEachRow` writer output for an accepted record (`internal/typelayer`'s `IngestWith`), `DEFAULT`s already filled in, re-encoded by the envelope's JSON encoding without changing any value — not a value WaveHouse encodes itself. - **claims.go**, **assign.go** — `ClaimShards` wraps a `Sharded` queue so an ingest process consumes only its share of the units, one process per unit at a time. Each process holds one membership lease, `ingest.m` for the lowest free `j` under the number of configured units (at most 64, read 16 at a time each tick), and every tick (2s) reads which slots are live (`coord.Observer.Held`); `assignUnits` gives every unit to a live slot by rendezvous hashing capped at ⌈units/slots⌉, the same result in every process for the same view; the extras are assigned apart from the configured units, so processes whose lists of extras differ (each refreshes it every five minutes) still agree on every configured unit. A unit no longer assigned halts (`mq.Halter`: stops fetching, keeps its pin, returns once what it fetched reached the worker), waits (bounded by the worker's 60s ack wait, the shortest `ack_wait` a durable may have) for the rows it delivered to settle (`Message.OnSettled`), then releases its pin (`mq.Releaser`). `Halt` on the claiming consumer ends the ticks and halts every unit at once; the worker calls it before its final flush, and the claims' stop releases the units and resigns the membership lease after it. A unit taken over from a slot that is no longer live is reset (`ResetOrphaned`) before it is bound — waiting, unbound, while the broker answers `ErrUnitHeld` (the dead owner's pin has not lapsed, or the unit was active within its pinned TTL), up to 30s — so the dead owner's unacked rows come back at once; on a process's first tick only if it is the one live member (a restart after a clean stop; after a crash the dead run's lease still counts as live for a lease duration). Each unit's rows delivered and unsettled are capped at its share of the worker's 10,000 (`ClaimConfig.MaxHeld`): an even share over the units assigned (`unitShare`, recomputed each tick and read by the broker before every fetch through `ConsumerConfig.MaxHeld`), never under 1,000, two batches; at its share a unit fetches only what keeps its pin, and one stuck unit never takes another's. A row stops counting at its first ack or nak attempt (`Message.OnSettled` fires whether or not the broker confirms), or after `ack_wait` if the worker never settles it, since the broker redelivers it then as a new row. A configured unit whose delivery ends, or whose durable or stream is gone or whose durable no longer fits (`mq.ErrConsumerNotFound`, `mq.ErrConsumerMismatch`) when it is bound, fails the worker; any other bind failure is retried next tick, logged as an error once it has lasted a minute, and an extra's end is logged. The lowest live slot counts the units with rows and no owner (`Sharded.Unowned`) for `wavehouse_ingest_shards_unowned`. - **sweeper.go** — `Sweeper` implements the Active Sweeper pattern. It runs every minute and asks the MQ (`mq.Purger.PurgeAcked`) to drop the ingest events that are **both** ACKed by the buffer consumer (written to ClickHouse) **and** older than the gap window (re-read every sweep: each tenant's own `stream.gap_window_minutes`, a rejected tenant's as its folder last had it (unbounded for one rejected since boot) — `internal/app`'s `gapWindows` — and none for a removed tenant). Finding the purge point is `internal/mq`'s (`purge.go`). diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index d78a3407..62998fec 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -59,7 +59,7 @@ flowchart LR Note the embedded broker's stream is **dual-use**: it is both the durable buffer feeding the worker and the replay buffer that SSE clients gap-fill from. That is why a custom sweeper exists there instead of plain work-queue auto-deletion; `mq.backend: nats` splits the two roles instead, with work-queue partitions and a separate history stream (see [Scaling out](#scaling-to-multiple-instances)). :::note[Omitted columns take their real DEFAULT, not `null`] -The batch that reaches `insertToClickHouse` is not assembled from the request body — it is the bytes `IngestWith` returned for each accepted record, produced by ClickHouse's own writer. An omitted column's `DEFAULT` (or the type's default — `NULL` on a `Nullable` column with none) was evaluated before that line existed, so a `Nullable(T) DEFAULT …` column takes its default exactly as an `INSERT` naming fewer columns would. Verified on ClickHouse 26.8. +The batch that reaches `insertToClickHouse` is not assembled from the request body — it is each accepted record's row as `IngestWith` returned it, ClickHouse's own writer output, carried through the envelope without changing any value. An omitted column's `DEFAULT` (or the type's default — `NULL` on a `Nullable` column with none) was evaluated before that line existed, so a `Nullable(T) DEFAULT …` column takes its default exactly as an `INSERT` naming fewer columns would. Verified on ClickHouse 26.8. ::: :::note[Insert settings pinned] From e3c3f148d71e0733ed2d1f3722d571d142822327 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 12:38:13 -0400 Subject: [PATCH 63/70] docs: a table whose schema cannot compile is unavailable on its own Co-Authored-By: Claude Opus 5.5 --- docs/src/content/docs/access-control.mdx | 2 +- docs/src/content/docs/api.md | 2 +- docs/src/content/docs/architecture.md | 9 +++++---- docs/src/content/docs/deployment.md | 2 +- 4 files changed, 8 insertions(+), 7 deletions(-) diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 4679b642..30dec5d6 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -367,7 +367,7 @@ The same policy drives every data path, but not every field is meaningful on eve :::caution[Live streams enforce column and row policy, but not resource limits] SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause on a server running that line's default settings, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72); on an integer column such a claim is a `filter` — and also when the event's parse (`Table.ParseRow`) is refused as a whole, which says nothing about the row), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with — only that tenant's rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72); on an integer column such a claim is a `filter` — and also when the event's parse (`Table.ParseRow`) is refused as a whole, which says nothing about the row), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with, or the table's schema could not be compiled (that table alone; logged as `chtypes could not compile table schema`) — only that tenant's (or that table's) rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 8511e0a8..b7d587a5 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -422,7 +422,7 @@ A `200` is returned whenever the body was read and the records were processed | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now; `Retry-After: 5`. Nothing in the window being reserved was published; the windows before it were, and keep their ids | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | A record's dedupe id is held by another request still being published; includes `Retry-After` (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). Nothing in that record's window was published; the windows before it were | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet; `Retry-After: 5`. Decided before the body is read | -| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet, or its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the zone this process opened that line with — the cause is in the server log, not the body. Decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | +| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet, or its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the zone this process opened that line with, or the table could not be compiled — the cause is in the server log, not the body. Decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index edeefc15..5cf1a6ca 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -85,7 +85,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`sql_classify.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. -- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (brackets that do not balance are a whole-request `400 invalid json: unterminated json array`, anything but whitespace after the closing `]` a whole-request `400 invalid json: content after the closing ']' of the json array`, and a `…WithNames` header naming an unknown or repeated column a whole-request `400 clickhouse.rejected` with code 117, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant the engine cannot answer for (no artifact for its ClickHouse line, or a server time zone that differs from the one this process opened that line with) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read, without affecting other tenants. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's duplicate window (two minutes on the embedded broker) — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). +- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (brackets that do not balance are a whole-request `400 invalid json: unterminated json array`, anything but whitespace after the closing `]` a whole-request `400 invalid json: content after the closing ']' of the json array`, and a `…WithNames` header naming an unknown or repeated column a whole-request `400 clickhouse.rejected` with code 117, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant or table the engine cannot answer for (the tenant not bound yet, no artifact for its ClickHouse line, a server time zone that differs from the one this process opened that line with, or a table schema the parser could not compile) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read, without affecting other tenants. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's duplicate window (two minutes on the embedded broker) — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` and `/` left unescaped via `output_format_json_escape_forward_slashes=0`, the same spellings the structured-query path and the SSE wire use, so a timestamp and a `/` read the same on every surface. - **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `output_format_json_named_tuples_as_objects=1`, `output_format_json_escape_forward_slashes=0`, `date_time_output_format=iso`, so a `/` is not escaped and a timestamp is RFC 3339 in UTC, both as on the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so `chconn.Classify` and `writeCHError` apply as on every other ClickHouse path, and a response past the 64 MiB cap is `clickhouse.response_too_large`. The read paths hold at most the tenant's pool size in HTTP connections — its `max_open_conns`, or the largest among the tenants sharing its connection tuple (address, database, user, password, TLS) — one cap per tuple, like the native pool; tenants on one server with a different database, user, password or TLS settings each have their own. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). @@ -279,9 +279,10 @@ Client POST /v1/ingest?table={table} (413 before any record is processed, so nothing is published) → Compile the ROLE's schema for the request's tenant: the columns it may insert, plus a DEFAULT per _eq check column carrying the claim (cached per - generation+shape). A tenant the engine cannot answer for (no artifact for - its ClickHouse line, or a server time zone that differs from the one this - process opened that line with) → 503 + Retry-After: 5, before the body is read + generation+shape). A tenant or table the engine cannot answer for (not bound + yet, no artifact for its ClickHouse line, a server time zone that differs + from the one this process opened that line with, or a table schema the + parser could not compile) → 503 + Retry-After: 5, before the body is read → Validate the whole body through chtypes (internal/typelayer), one call per request: ClickHouse's own parser type-checks, coerces, and fills DEFAULTs (including now()) per record. A rejected record carries ClickHouse's real diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 722379a2..45dee655 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -116,7 +116,7 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ **Which processes load it.** Only processes with the `api` role — the ones that serve ingest and the stream. A process that runs only the ingest worker or the sweeper loads no artifact and boots without one installed. An API process refuses to start when no artifact is installed at all. -**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with the cause in the server log; run such tenants in a separate process, or align the servers' `timezone` setting. The health probes stay green through either — `/livez` and `/readyz` do not read the type layer — so watch for the `ERROR` log line `chtypes cannot serve this tenant`, the ingest `503`s, and `wavehouse_sse_rows_withheld_total{reason="unavailable"}`. See [API → Ingest error responses](/api#post-v1ingesttabletable--ingest-data) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). +**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with the cause in the server log; run such tenants in a separate process, or align the servers' `timezone` setting. The health probes stay green through either — `/livez` and `/readyz` do not read the type layer — so watch for the `ERROR` log line `chtypes cannot serve this tenant`, the ingest `503`s, and `wavehouse_sse_rows_withheld_total{reason="unavailable"}`. See [API → Ingest error responses](/api#post-v1ingesttabletable--ingest-data) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). A single table whose schema the parser cannot compile is refused the same way, on its own — that table's ingest answers `503` and its row-filtered stream rows are withheld as `unavailable` — and is logged as `chtypes could not compile table schema` rather than `chtypes cannot serve this tenant`. **Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse never fetches an artifact itself: a line with no installed artifact makes the tenants on it unavailable at their schema refresh, and boot is refused only when no artifact is installed at all, or when an explicit `clickhouse.chtypes_registry` does not exist or cannot be read. One exception: the SDK honors `CHTYPES_AUTOFETCH=1` from the environment over WaveHouse's setting, and would then download a missing line (hundreds of MB) inside a schema refresh — leave it unset. From 9d8f95ac1cb99565e40701fa3da6ec52930c9e2f Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 13:00:00 -0400 Subject: [PATCH 64/70] docs: every copy of the unavailable claim names all four causes A tenant is unavailable when it is not bound yet, its line has no installed artifact, or its server zone differs from the one the process opened that line with; a table whose schema does not compile is unavailable alone. Several copies named only the tenant-wide causes or said "only that tenant is refused". Deployment -> chtypes artifacts now holds the full list, how each cause recovers (checked again at every schema refresh; a failed table is compiled again at every bind) and the three ERROR log lines to watch. The API tables, the SSE note, access control, the architecture page, the SDK reference, AGENTS.md, the changelog and the Go doc comments state the same set or point there. The ingest diagram moves the 503 to its own step before the body is read, where the code checks it. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 4 ++-- CHANGELOG.md | 2 +- docs/src/content/docs/access-control.mdx | 2 +- docs/src/content/docs/api.md | 10 ++++---- docs/src/content/docs/architecture.md | 17 +++++++------- docs/src/content/docs/deployment.md | 4 +++- docs/src/content/docs/development.md | 2 +- docs/src/content/docs/sdk/reference.md | 2 +- internal/api/ingest.go | 30 +++++++++++++----------- internal/app/wire.go | 6 ++--- internal/stream/metrics.go | 15 ++++++------ internal/stream/roweval.go | 10 ++++---- internal/typelayer/errors.go | 7 +++--- internal/typelayer/typelayer.go | 16 +++++++------ 14 files changed, 69 insertions(+), 58 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 57bb177d..f4b649cb 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -47,7 +47,7 @@ Twenty-one internal packages under `internal/` (plus `internal/testutil/` for sh - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes - **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row via `typelayer`, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`typelayer/`** — the only package (with its test helper `typelayertest`) that imports `github.com/wave-rf/chtypes/go/chtypes`, by convention: no depguard rule enforces it. One process-wide `Engine` wraps one `chtypes.Registry`, opened lazily from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from that tenant's `discovery` refresh) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema, and `Forget` releases a tenant that is no longer served. A tenant with no matching artifact, or whose server time zone differs from the zone this process already opened that ClickHouse line with, is unavailable on its own while every other tenant keeps working. `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns (a denied column stays as `MATERIALIZED` of its default, so naming it is code 117), plus a `DEFAULT ''` per `_eq` check column, and any `EPHEMERAL` column the role may write that a `DEFAULT` reads and no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` expression reads — which is how column policy and auto-inject are answered with no Go-side record inspection; a role shape's handle pool grows to `min(GOMAXPROCS, 4)` under load. A role schema that cannot compile is `500` with `retryable:false`. Test helpers live in `internal/typelayer/typelayertest` (`TestEngine`, `SkipWithoutArtifact`). `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) +- **`typelayer/`** — the only package (with its test helper `typelayertest`) that imports `github.com/wave-rf/chtypes/go/chtypes`, by convention: no depguard rule enforces it. One process-wide `Engine` wraps one `chtypes.Registry`, opened at boot from a registry directory (`clickhouse.chtypes_registry` / `WH_CHTYPES_REGISTRY`) with each ClickHouse line's library opened lazily by the first tenant bound to it, and built only by a process running the `api` role. Each tenant has its own table set: `Engine.Bind` (fired from every successful `discovery` refresh of that tenant) resolves the artifact matching the server's minor line — no nearest-version fallback — and recompiles a `Table` handle per changed schema and per table whose last compile failed, and `Forget` releases a tenant that is no longer served. A tenant not bound yet, one with no matching artifact, or one whose server time zone differs from the zone this process already opened that ClickHouse line with is unavailable on its own, and so is a table whose schema chtypes could not compile (that table alone), while every other tenant keeps working; the causes, their log lines and how each recovers are in [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). `Engine.RoleTable` compiles and caches a role's own schema — its insertable columns (a denied column stays as `MATERIALIZED` of its default, so naming it is code 117), plus a `DEFAULT ''` per `_eq` check column, and any `EPHEMERAL` column the role may write that a `DEFAULT` reads and no `MATERIALIZED`, `ALIAS` or `EPHEMERAL` expression reads — which is how column policy and auto-inject are answered with no Go-side record inspection; a role shape's handle pool grows to `min(GOMAXPROCS, 4)` under load. A role schema that cannot compile is `500` with `retryable:false`. Test helpers live in `internal/typelayer/typelayertest` (`TestEngine`, `SkipWithoutArtifact`). `Table.IngestWith(format, opts, body, checks...)` runs one request body through ClickHouse's own reader (`JSONEachRow`/`CSV`/`TSV`/`CSVWithNames`/`TSVWithNames`), returning a verdict per input record (accepted / rejected with ClickHouse's code and message / declined) plus the accepted rows as `JSONCompactEachRow` bytes; the role's insert checks run in that same parse as a compiled row filter (parse outcome first, then the check verdict), and `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event — one compiled-filter mechanism, values bound as `{p:String}` (Key Design Decision #21) - **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -74,7 +74,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention; `/readyz` pings every open ClickHouse pool at once and is ready at the first answer, 503 naming each when none answers); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. 19. **Timestamps agree on the wire by construction, not by rewriting** — the NATS/SSE `row` for `DateTime`/`DateTime64` columns is the exact bytes ClickHouse's own writer produced for the stored record (`typelayer`'s `Table.IngestWith`, via chtypes), as RFC 3339 in UTC at the column's scale (e.g. `"2026-06-21T04:00:00.123Z"`), because `date_time_output_format=iso` is pinned on both surfaces (as is `output_format_json_escape_forward_slashes=0`, so neither escapes `/`) — the ingest export settings (`internal/typelayer/ingest.go`) and the read path's `chReadSettingsFixed` (`internal/api/clickhouse_http.go`), which must change together with a `chRendering` bump — so there is no separate WaveHouse rewrite step to keep in sync and live and query reads can't drift on spelling *or* instant (#372). The column's or server's zone affects only how zone-less *input* is read, never the rendered spelling. Preserve when touching `internal/typelayer`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `typelayer/` + §Ingest Path; the wire shape lives in api.md §Timestamp rendering. 20. **Sealed MQ boundary** — only `internal/mq` imports NATS/JetStream (`github.com/nats-io/…`), enforced by the `depguard` rule in `.golangci.yml`, so `make lint` fails on a leak in every package it builds (the `integration`-tagged files under `tests/` are outside lint's build context — keep them clean by convention, through `mq.Broker`). A test outside `internal/mq` that needs a real NATS server goes through `internal/mq/natstest`, which stands one up from the shipped `deployments/nats` files and hands back a URL and passwords, never a NATS type. The boundary is semantic as well: everything else addresses events by `mq.Topic` and states intent through mq-owned interfaces (`Publisher`, `Consumer`, `DeadLetterer`, `Purger`, `Replayer`, …), and never builds a subject, names a stream, or reasons in sequences — so a subject, stream, or broker change lands in one package ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 4; story 5's tenant token landed there alone — `Topic.Tenant`, first in every subject). Don't add a raw accessor (`JetStream()`, `NatsConn()`, `GetServer()`) back, and don't hand-build `"ingest."`/`"dlq."` subjects outside `internal/mq` — widen the mq surface with an intent-level method instead. -21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` (with its test helper `typelayertest`) is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is unavailable **individually**: ingest answers `503` (generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds every row of a role that has a row `filter` with reason `unavailable`, and other tenants keep working. Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and, on Linux, glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. +21. **ClickHouse's own parser validates ingest and evaluates row-level security, in-process (security)** — `internal/typelayer` (with its test helper `typelayertest`) is the only importer of `github.com/wave-rf/chtypes/go/chtypes`, a per-ClickHouse-minor-version shared library loaded via `dlopen` and matched to the connected server's line with **no nearest-version fallback**, and only by a process running the `api` role. A tenant not bound yet, one whose ClickHouse line has no installed artifact, or one whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line) is unavailable **individually**, and so is a table whose schema chtypes could not compile (that table alone): ingest answers `503` (`Retry-After: 5`, generic body `ingest validation is unavailable`, the cause in the log only), the stream withholds that tenant's (or that table's) rows from every role that has a row `filter`, with reason `unavailable`, and other tenants keep working — recovery and log lines in [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Ingest validation, type coercion, and `DEFAULT` substitution run ClickHouse's real parser over the whole request body in one call, so a rejection carries ClickHouse's own error code (`exception_code`, beside the string `code` class) and message instead of a WaveHouse-authored sentence — an unknown column, a computed-only column and **a column the role may not write** are all **117**, because column policy is answered by compiling the role its own schema (`Engine.RoleTable`) rather than by walking a decoded record; a record the engine cannot answer for is **declined** (`422`), distinct from and never conflated with a data rejection (`400`). Predicates — a role's row `filter` and its insert `check` alike — compile through chtypes with every bound value a `{p:String}` parameter, never interpolated, and are evaluated the way the server's `WHERE` clause would evaluate them, for every column type. Only a definite true admits; error, decline, schema drift, or an unavailable engine withhold (fail closed), each counted separately in `wavehouse_sse_rows_withheld_total{table,role,reason}` (`filter`, `error`, `decline`, `unavailable`, `drift`). A reader whose filter uses a column the inserting role cannot write, or a `MATERIALIZED` column, is declined every such row on the stream, though `/v1/query` returns it. Consequence: the binary requires cgo (dlopen only, no static link to the artifact) and, on Linux, glibc, so supported platforms are Linux amd64/arm64 and macOS arm64 — see [Deployment → chtypes artifacts](docs/src/content/docs/deployment.md#chtypes-artifacts). Preserve when touching `internal/typelayer`, ingest, or the stream row-filter; change the artifact-matching or fail-closed behavior only with a security review. Detail: architecture.md § `typelayer/`. ## Code Conventions diff --git a/CHANGELOG.md b/CHANGELOG.md index 7754ea00..919154af 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -63,7 +63,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `project_name`, `git.ignore_tags` and `builds:`, and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes` with `--notes-start-tag` set to the previous `v*` tag, `git describe --tags --abbrev=0 --match 'v*' "${tag}^"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. -- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser in one call per request, as-is except that a JSON array is re-framed in place to one element per line so one bad element cannot lose the rest, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable (`33` for a record cut off mid-object), `117` unknown field, otherwise the type's own code (`72`, `41`, `38`, `69`, `376`, `467`, `691`, `675` …); an out-of-range integer wraps exactly as ClickHouse's own `INSERT` does (`256` → `0` in a `UInt8`) — so a malformed record is no longer the whole-request `400 {"error":"invalid json"}` (a JSON array whose brackets do not balance, or with content after its `]`, is still a whole-request `400 {"error":"invalid json: …"}`); a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — published rows, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date` and `Date32` keep their `"2026-06-21"` form on the stream, and on `/v1/query` and pipes they lose the `T00:00:00Z` the native path added — see the structured-query entry) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant whose ClickHouse line has no installed artifact, or whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), is refused on its own — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable` — while every other tenant keeps working. Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A dedupe id is read from the exported row: a `null` cell or an empty string is missing (so is an omitted `String` id with no `DEFAULT`), and a numeric id column cannot tell an omitted `0` from a supplied one. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser in one call per request, as-is except that a JSON array is re-framed in place to one element per line so one bad element cannot lose the rest, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable (`33` for a record cut off mid-object), `117` unknown field, otherwise the type's own code (`72`, `41`, `38`, `69`, `376`, `467`, `691`, `675` …); an out-of-range integer wraps exactly as ClickHouse's own `INSERT` does (`256` → `0` in a `UInt8`) — so a malformed record is no longer the whole-request `400 {"error":"invalid json"}` (a JSON array whose brackets do not balance, or with content after its `]`, is still a whole-request `400 {"error":"invalid json: …"}`); a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — published rows, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date` and `Date32` keep their `"2026-06-21"` form on the stream, and on `/v1/query` and pipes they lose the `T00:00:00Z` the native path added — see the structured-query entry) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant not bound yet, one whose ClickHouse line has no installed artifact, or one whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line) is refused on its own, and so is a table whose schema chtypes could not compile (that table alone) — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable` — while every other tenant keeps working. Each cause is checked again at every schema refresh and clears at the first one after it is fixed ([Deployment → chtypes artifacts](https://wavehouse.dev/deployment#chtypes-artifacts)). Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A dedupe id is read from the exported row: a `null` cell or an empty string is missing (so is an omitted `String` id with no `DEFAULT`), and a numeric id column cannot tell an omitted `0` from a supplied one. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. - **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, instead of walking a decoded record's keys. A column the role may not write stays in that schema as `MATERIALIZED` of its default, so naming it is refused while expressions that read it keep working and the stored row holds the table's default. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a non-`Nullable` checked column behaves exactly like omitting it (on a `Nullable` one it stores `NULL`, which fails the check). An `_eq` check also covers a column the role may not otherwise write: a record that omits it is filled with the required value and published, one that supplies exactly that value is accepted (**0.1.0 answered `403 column "x" not allowed for insert` to the correct value**), and any other value is `403 check failed for column "x"`. A role whose schema cannot be compiled this way, or that may write no column of the table, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After` (the cause is logged once a minute), rather than a `503` that a retry could not fix; a policy `check` on an `EPHEMERAL` column is still `403`. diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 30dec5d6..a7103df2 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -367,7 +367,7 @@ The same policy drives every data path, but not every field is meaningful on eve :::caution[Live streams enforce column and row policy, but not resource limits] SSE subscribers are checked for table-level `select` permission, have denied columns stripped from each event, and receive only the rows their role's row `filter` admits, evaluated per subscriber against their JWT claims. Row-level security on the stream is evaluated by the **same engine as the server's own `WHERE` clause**: each event is parsed once (`internal/typelayer.Table.ParseRow`) and each subscriber's resolved predicates are compiled — claim values bound as `{p:String}` parameters, never interpolated — and evaluated against it (`Row.Visible`). Because this is ClickHouse's own parsing and comparison, every column type compares exactly as it would in a real `WHERE` clause on a server running that line's default settings, and there is no per-type comparison table to reconcile with the server. Predicates are evaluated against the **full ingested event**, so a filter may key on a column the role cannot `select`. -Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72); on an integer column such a claim is a `filter` — and also when the event's parse (`Table.ParseRow`) is refused as a whole, which says nothing about the row), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the one this process already opened that line with, or the table's schema could not be compiled (that table alone; logged as `chtypes could not compile table schema`) — only that tenant's (or that table's) rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts)), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. +Only a **definite true** admits a row. Everything else withholds, and `wavehouse_sse_rows_withheld_total{table,role,reason}` counts each cause separately so a quiet stream's reason is visible rather than guessed: `filter` (a definite non-match), `error` (the predicate errored on this row — no supertype between constant and column, or a constant a non-integer column's type cannot read, such as `abc` on a `Decimal` (ClickHouse's code 53) or a `Float` (code 72); on an integer column such a claim is a `filter` — and also when the event's parse (`Table.ParseRow`) is refused as a whole, which says nothing about the row), `decline` (the engine would not answer, which is what a row it cannot read gets, or the filter reads a column the event does not carry, below), `unavailable` (the tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, its server time zone differs from the one this process already opened that line with, or the table's schema could not be compiled (that table alone; logged as `chtypes could not compile table schema`) — only that tenant's (or that table's) rows are withheld, and only from a role with a row `filter`; see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for how each recovers), and `drift` (the event's column list and the live table disagree after a mid-stream `ALTER`). A row published by a column-restricted role carries only the columns that role may write, and the stream evaluates it against that column list. A filter over a column the published row does not carry is **declined** (`decline`) for that subscriber rather than treated as a mismatch on an absent value: that is a column the inserting role cannot write, a `MATERIALIZED` or `ALIAS` column (computed by ClickHouse, never part of a published row), or one the table no longer has. The reader never receives those rows over the stream, although `/v1/query` returns them — a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row (`now()` or `rand()` would land differently), so a verdict on the stream's copy could admit a row the query path excludes. The remedy is a filter on a column every inserting role writes. A filter the engine will not compile at all, which a policy WaveHouse rendered is not expected to produce, withholds every row for that role (`decline`) until the filter or the table changes; the failure is logged once per parser handle and schema generation, not per event. A constant the column type cannot read is not that case: it compiles, and errors on each row as above. Two edges follow from the stream evaluating the **ingested event** rather than re-reading the stored row. An **omitted `DEFAULT` column is not a problem case**: chtypes evaluated the `DEFAULT` before publish, so the event carries the real value, and the [`check` + `filter` pairing](#insert-checks) works — an `_eq` insert check stamps its claim into any payload that omits the column *before* publish, so the streamed event carries it and the matching row filter evaluates normally. What remains is the other direction: an event whose insert later **fails outright** at ClickHouse (a value ClickHouse rejects, which the DLQ parks) was already streamed to whichever subscribers the filter admitted, and its row never becomes queryable. A ClickHouse outage only delays the row, which is retried until it inserts. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index b7d587a5..06f587db 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -281,7 +281,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | | 422 | `{"error":"validation engine declined: "}` | **Per-record.** chtypes did not answer for the record. That can be the shape (the artifact declined it), or a body it could not read as a whole — a record cut off right after a key's `:` or an opening quote, or a single-line array declared NDJSON holding a bad element — in which case every record in the body, good ones included, gets this answer and nothing is published. A `check` clause that could not be evaluated lands here too (`validation engine declined: the insert check for column "x" could not be evaluated`) | -| 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither the record's fault nor an unavailable tenant; logged. Nothing was published | +| 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither the record's fault nor an unavailable tenant or table (the `503` below); logged. Nothing was published | | 500 | `{"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` | The role's insert permissions do not compile against the table (for example, the role may write no column of it). It persists until the policy or the table changes, so it carries no `Retry-After` and must not be retried; the cause is in the server log | | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now: it is not open (for example, it failed to open on a reload), or a DynamoDB table is throttling, timing out or unreachable; `Retry-After: 5`. Nothing was published, so the retry is safe | @@ -290,7 +290,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure, for that tenant alone) or not open (see [Message Queue](/settings-directory#message-queue)). Under [`mq.backend: nats`](/deployment#external-nats), the table's partition stream is full, which refuses every table in it, or the tenant's table holds as many unwritten rows as the stream allows one subject. Response includes `Retry-After: 30` header. With dedupe on, the record's id is given back, so the retry is published rather than reported as a duplicate. | | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`, a transient broker failure, not a full queue). Only under [`mq.backend: nats`](/deployment#external-nats), including a partition stream the operator deleted; the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the record's id is left to lapse rather than given back, so a retry cannot land as a second copy; `Retry-After` is that lease, rounded up to whole seconds, when dedupe was on for the record, else the flat `Retry-After: 5`. | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | Dedupe is on and another request carrying the same id is still being published — usually a client's timeout-retry racing its own original. Its outcome decides whether this record is a duplicate, so retry after the `Retry-After` header (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). | -| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet (no completed schema refresh), a table could not be compiled, its ClickHouse line has no installed chtypes artifact for its server version, or its server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line). The body is generic on purpose — the cause, with zone names and artifact search paths, goes to the server log only. Only that tenant is refused — every other tenant keeps working — and it is decided before the body is read, with `Retry-After: 5`. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) | +| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet (no completed schema refresh), its ClickHouse line has no installed chtypes artifact for its server version, its server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), or the table's schema could not be compiled. The body is generic on purpose — the cause, with zone names and artifact search paths, goes to the server log only. Only that tenant (or, for a table that could not be compiled, only that table) is refused — every other tenant keeps working — and it is decided before the body is read, with `Retry-After: 5`. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for how each cause recovers | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | **curl example:** @@ -414,7 +414,7 @@ A `200` is returned whenever the body was read and the records were processed | 404 | `{"error":"unknown table: ..."}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | -| 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither a record's fault nor an unavailable tenant; logged. Nothing was published | +| 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither a record's fault nor an unavailable tenant or table (the `503` below); logged. Nothing was published | | 500 | `{"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` | The role's insert permissions do not compile against the table (for example, the role may write no column of it). It persists until the policy or the table changes, so it carries no `Retry-After` and must not be retried; the cause is in the server log | | 500 | `{"error":"publish failed"}` / `{"error":"dedupe failed"}` | Message-queue or dedup-backend failure mid-batch, other than a full queue or an unreachable broker (below). After a publish failure the records before it keep their ids, so a whole-batch retry reports those as duplicates; the failing record's id is left to lapse as on the single-object path, and the rest of its window's ids are given back | | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure) or not open, mid-batch; includes `Retry-After: 30`. The records before the refused one keep their ids, and its id and the rest of its window's are given back | @@ -422,7 +422,7 @@ A `200` is returned whenever the body was read and the records were processed | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now; `Retry-After: 5`. Nothing in the window being reserved was published; the windows before it were, and keep their ids | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | A record's dedupe id is held by another request still being published; includes `Retry-After` (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). Nothing in that record's window was published; the windows before it were | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet; `Retry-After: 5`. Decided before the body is read | -| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet, or its ClickHouse line has no installed chtypes artifact, or its server time zone differs from the zone this process opened that line with, or the table could not be compiled — the cause is in the server log, not the body. Decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | +| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, its server time zone differs from the zone this process opened that line with, or the table's schema could not be compiled (that table alone) — the cause is in the server log, not the body. Decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] @@ -689,7 +689,7 @@ Each SSE connection is bound to a single `?table=`; to consume multiple tables, Values of `DateTime`/`DateTime64` columns inside `row`, nested ones included, are ClickHouse's own rendering of the stored value — the exact bytes chtypes' `RowsExport` produced for that record (see [Timestamp rendering](#timestamp-rendering)), not a WaveHouse rewrite — so a live event and a `/v1/query` read of the same row agree on spelling **by construction**, with no separate canonicalization step to keep in sync ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). A column declared with a non-UTC zone is still rendered in UTC (the `Z` form), so the strings compare as the instants do. -**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause on a server running that line's default settings — there, a connection is never delivered a row the query path would hide for that role (a server profile that changes a comparison setting can split the two; see [JWT claim templating](/access-control#jwt-claim-templating)), and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant whose ClickHouse line has no installed artifact is unavailable on its own: a stream whose role has a row `filter` withholds that tenant's rows with reason `unavailable`, while other tenants' streams (and roles with no row filter) are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant's line is not served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` or `ALIAS` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert ClickHouse later rejects and parks on the dead-letter queue — a record the in-process parse accepted that the insert did not (a schema change between publish and insert, for instance), since an outage only delays a row and never drops it — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), but an expired token or changed claims take effect only when the client reconnects. +**Note:** When access control policies are active, streamed events are filtered per the caller's role: tables without `select` permission are skipped, denied columns are removed from each event, and the role's [row-level `filter`](/access-control#row-level-security) is compiled and evaluated per subscriber against the caller's JWT claims — supplied by the connection's token (the `Authorization` header, or the `?token=` fallback above), with replayed gap-fill events filtered the same way. This runs through the same in-process ClickHouse parser (chtypes) that validates ingest, so every column type compares exactly as it would in the query path's `WHERE` clause on a server running that line's default settings — there, a connection is never delivered a row the query path would hide for that role (a server profile that changes a comparison setting can split the two; see [JWT claim templating](/access-control#jwt-claim-templating)), and a predicate that can't compile or evaluate withholds the row instead of guessing (see [the enforcement caution](/access-control#where-each-rule-is-enforced) for the fail-closed reasons). A tenant or table the type layer cannot serve is unavailable on its own — the tenant not bound yet, no installed artifact for its ClickHouse line, a server time zone that differs from the one this process opened that line with, or a table schema chtypes could not compile (that table alone); see [Deployment → chtypes artifacts](/deployment#chtypes-artifacts). A stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable`, while other tenants' streams (and roles with no row filter) are unaffected. A withheld row is counted by `wavehouse_sse_rows_withheld_total{table,role,reason}`; `reason` is `filter` (the predicate answered false), `error` (it failed to evaluate), `decline` (the engine cannot answer for the row), `unavailable` (the tenant or the table cannot be served, above) or `drift` (the event names a column the table no longer has, as after a schema change). A role's row `filter` over a column the inserting role cannot write, or over a `MATERIALIZED` or `ALIAS` column, never streams to that reader: a published row carries only the columns its inserting role wrote, and a `DEFAULT` or `MATERIALIZED` value is computed again when ClickHouse stores the row, so the stream declines the row rather than guess — `/v1/query` still returns it. The residual payload-vs-stored case is an event whose insert ClickHouse later rejects and parks on the dead-letter queue — a record the in-process parse accepted that the insert did not (a schema change between publish and insert, for instance), since an outage only delays a row and never drops it — which the caution documents. The connection's claims are captured once, when the stream is established — a policy change applies from the next event, replayed or live (a gap-fill re-reads the policy per event too), but an expired token or changed claims take effect only when the client reconnects. **CORS:** `/v1/stream` honors the request's tenant's `cors.allowed_origins` allowlist (settings directory) like every endpoint — the preflight included, which a browser sends without `X-Tenant-ID`, so over [a nested settings directory](/deployment#multi-tenant-deployments) the fronting proxy has to set the header on the `OPTIONS` too. Note that a **header-authenticated stream preflights before it connects** — `Authorization` is not CORS-safelisted — where a bare `EventSource` never preflighted at all: its request is not a `fetch()`, so Fetch's unsafe-request flag is never set and `Last-Event-ID` rides on the plain `GET`. Both headers are allow-listed, so an allowed origin connects *and* resumes cross-origin. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 5cf1a6ca..623dd32c 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -106,7 +106,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)) lives next to the keepalive primitives it shares. One abstraction per file. -- **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row. Visibility itself is decided by `internal/typelayer` (`Table.ParseRow` once per event, on the event's tenant's table set, then `Row.Visible` per subscriber) — the same ClickHouse parsing and comparison semantics the server's own `WHERE` clause applies, for every column type, rather than a hand-written per-type comparator; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (one the inserting role cannot write, or a `MATERIALIZED`/`ALIAS` column; reason `decline`, though `/v1/query` still returns the row), a schema drift between the event and the live table, or an engine that is unavailable for that tenant all withhold the row rather than guessing. Each row withheld this way increments `wavehouse_sse_rows_withheld_total{table,role,reason}`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and, under a row filter, preparing one parsed row per replayed event (closed at once, since a gap-fill can run for thousands of events). `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. +- **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row. Visibility itself is decided by `internal/typelayer` (`Table.ParseRow` once per event, on the event's tenant's table set, then `Row.Visible` per subscriber) — the same ClickHouse parsing and comparison semantics the server's own `WHERE` clause applies, for every column type, rather than a hand-written per-type comparator; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (one the inserting role cannot write, or a `MATERIALIZED`/`ALIAS` column; reason `decline`, though `/v1/query` still returns the row), a schema drift between the event and the live table, or an engine that is unavailable for that tenant or table (reason `unavailable`; the causes are in the `typelayer/` section below) all withhold the row rather than guessing. Each row withheld this way increments `wavehouse_sse_rows_withheld_total{table,role,reason}`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and, under a row filter, preparing one parsed row per replayed event (closed at once, since a gap-fill can run for thousands of events). `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. - **roweval.go** — `RowEvaluator` / `RowView`, the one place stream row visibility under a role's row `filter` is decided: `Prepare` parses an event once through `internal/typelayer`, and the view answers per subscriber. A nil evaluator fails closed, and `WithheldReason` maps a prepare error to its `reason` label of `wavehouse_sse_rows_withheld_total` (`unavailable`, `drift` or `error`); the view's `Visible` supplies `filter`, `decline` and `error` (ClickHouse raising while evaluating the predicate over the row). - **subscriber.go** — `Subscriber`, the per-connection handle. It carries the connection's JWT claims, fixed at construction (`NewSubscriber(claims, metrics)`, no setter) — the claims the `Hub` resolves a role's row-level `filter` against, and immutability is what makes the fan-out's unsynchronized claims read race-free structurally. It owns a single ready-to-write outbound queue of `Frame`s (each tagged with its `kind`, so the handler labels the write where it happens): producers — the keepalive wheel and the event `Hub` — fan frames in with `Send` (non-blocking; a full queue drops, and `Send` itself counts the drop by frame kind, so no producer can forget to), and the handler drains `Frames()` to the client verbatim. The queue is sized for buffering live events (cap 64, up from the keepalive-only cap 1; #152 will make it a knob), and `Evict` closes its `Evicted()` channel, once, for the handler to end the stream: the `Hub`'s `Prune` does for a tenant no longer served, and the slow-consumer follow-up will for a wedged consumer. - **bucket.go** — `Bucket`, the reusable fan-out primitive: a concurrency-safe set of subscribers. `Push` fans one `Frame` to every member fire-and-forget — the keepalive wheel's ring is its only caller now that both `Hub` paths iterate `Snapshot`, since the schema announcement is per connection even where the projection is shared per role; `Snapshot` exposes the members so the event `Hub` can evaluate row visibility per subscriber before sending (drop counting lives in `Send` itself). The `Hub` holds one `Bucket` per `(topic, role)` so a projected frame is built once and sent to every member instead of re-projected per subscriber. @@ -169,12 +169,12 @@ The only package (with its test helper `typelayertest`) that imports `github.com Each tenant has its own table set inside that engine, bound from the tenant's own schema refresh and released when the tenant is no longer served. -- **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind. A tenant whose ClickHouse line has no installed artifact is unavailable on its own, and so is one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Either way the cause is recorded (an ingest client sees only a generic `503`; the cause goes to the log), every other tenant keeps working, and a later bind of the same tenant (a reload, a refresh that now agrees) clears it. +- **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant, on every successful refresh: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind or whose last compile failed, and every table when the tenant's library changes. Until its first bind a tenant is unavailable on its own (not bound yet). So is a tenant whose ClickHouse line has no installed artifact, and one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Those two are logged as `chtypes cannot serve this tenant`, and a later bind of the same tenant that resolves its line (a refresh after the artifact is installed, or once the server reports that line's zone) clears them. A table whose schema chtypes could not compile is unavailable on its own, logged as `chtypes could not compile table schema`, and is compiled again at every bind until it compiles. In every case the cause is recorded (an ingest client sees only a generic `503`; the cause goes to the log), and every other tenant keeps working, as do the tenant's other tables when only one table failed to compile. [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) has the operator's view: how each cause recovers and what to watch. - **`Engine.RoleTable(tenant, table, shape)`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column, including one the role may not otherwise write — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column stays in the schema as `MATERIALIZED` of its default, so naming it is ClickHouse's code 117 while expressions that read it still compile, and an absent check column takes the claim as its default while a supplied value is accepted only if it equals the claim. A role-shape pool starts at one handle and grows to `min(GOMAXPROCS, 4)` when every handle is busy; a base table's grows to `min(GOMAXPROCS, 8)`. - **`Table.IngestWith(format, opts, body, checks...)`** (`Ingest` is the same with no options) runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. - **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per handle (up to 4096 on each). Only a definite true admits; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (`decline`), schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. -- A tenant that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row of that tenant's tables from a role that has a row `filter`, with reason `unavailable`. -- A role whose schema cannot be compiled — or that may write no column of the table — is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`, and the cause is logged once a minute; it is not the `503` an unavailable tenant gets, since retrying cannot fix a policy or a table. A shape that fails only because an injected check value cannot be its column's default (`'abc'` on a `UInt64`) is retried without the defaults and served: a record omitting that column then fails the check (`403` on an integer column, `422` on another), and only a shape that does not compile even without them gets the `500`. +- A tenant or table that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row of that tenant's tables (or of that one table) from a role that has a row `filter`, with reason `unavailable`. +- A role whose schema cannot be compiled — or that may write no column of the table — is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`, and the cause is logged once a minute; it is not the `503` an unavailable tenant or table gets, since retrying cannot fix a policy or a table. A shape that fails only because an injected check value cannot be its column's default (`'abc'` on a `UInt64`) is retried without the defaults and served: a record omitting that column then fails the check (`403` on an integer column, `422` on another), and only a shape that does not compile even without them gets the `500`. - **`typelayertest`** (`internal/typelayer/typelayertest`) holds the test helpers other packages' tests use — `TestEngine`, `SkipWithoutArtifact`, and `RequireEnv` (the `WAVEHOUSE_TEST_REQUIRE_CHTYPES` switch that turns a skip into a failure) — which only test binaries link (`typelayer` itself never imports `testing`). The `typelayer` package's own tests use a small in-package copy, since they cannot import a package that imports them. See [API → Ingest](/api#post-v1ingesttabletable--ingest-data) for the ingest error-response shape and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced) for how predicates are compiled and evaluated. @@ -275,14 +275,15 @@ Client POST /v1/ingest?table={table} → Policy check: role allowed to insert into this table (before the body is parsed) → Resolve the declared Content-Type into the body's format (415 if absent, unsupported, or if declarations disagree; before the body is read) + → A tenant or table the engine cannot answer for (not bound yet, no artifact + for its ClickHouse line, a server time zone that differs from the one this + process opened that line with, or a table schema the parser could not + compile) → 503 + Retry-After: 5, before the body is read → Read the whole body into a pooled buffer, bounded by the 16 MiB cap (413 before any record is processed, so nothing is published) → Compile the ROLE's schema for the request's tenant: the columns it may insert, plus a DEFAULT per _eq check column carrying the claim (cached per - generation+shape). A tenant or table the engine cannot answer for (not bound - yet, no artifact for its ClickHouse line, a server time zone that differs - from the one this process opened that line with, or a table schema the - parser could not compile) → 503 + Retry-After: 5, before the body is read + generation+shape) → Validate the whole body through chtypes (internal/typelayer), one call per request: ClickHouse's own parser type-checks, coerces, and fills DEFAULTs (including now()) per record. A rejected record carries ClickHouse's real diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 45dee655..ff88a41e 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -116,7 +116,9 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ **Which processes load it.** Only processes with the `api` role — the ones that serve ingest and the stream. A process that runs only the ingest worker or the sweeper loads no artifact and boots without one installed. An API process refuses to start when no artifact is installed at all. -**What a mismatch does.** A tenant whose ClickHouse line has no installed artifact is refused on its own — ingest answers `503` (`Retry-After: 5`) and a stream whose role has a row `filter` withholds its rows with reason `unavailable` — while every other tenant keeps working; it recovers at the next schema refresh once an artifact is installed. The same holds for the server time zone: the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**. A tenant whose server reports a different zone from the one this process already opened that line with is refused the same way, with the cause in the server log; run such tenants in a separate process, or align the servers' `timezone` setting. The health probes stay green through either — `/livez` and `/readyz` do not read the type layer — so watch for the `ERROR` log line `chtypes cannot serve this tenant`, the ingest `503`s, and `wavehouse_sse_rows_withheld_total{reason="unavailable"}`. See [API → Ingest error responses](/api#post-v1ingesttabletable--ingest-data) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). A single table whose schema the parser cannot compile is refused the same way, on its own — that table's ingest answers `503` and its row-filtered stream rows are withheld as `unavailable` — and is logged as `chtypes could not compile table schema` rather than `chtypes cannot serve this tenant`. +**When a tenant or table is unavailable.** The type layer refuses a whole tenant, on its own, for three causes: the tenant is not bound yet (its schema discovery has not completed a refresh); its ClickHouse line has no installed artifact (or only one the SDK refuses to load, such as a truncated library); or its server reports a different time zone from the one this process already opened that line with — the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**, and the first tenant bound on a line fixes that zone. A fourth cause refuses one table and leaves the tenant's other tables working: a table whose schema the parser could not compile. In every case, ingest into that tenant (or that table) answers `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable`, decided before the body is read; a stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable`, while roles with no row filter are unaffected; and every other tenant keeps working. + +**How it recovers, and what to watch.** Every cause is checked again at each of the tenant's schema refreshes (`schema.refresh_interval`), and clears at the first refresh after the cause is gone. A tenant not bound yet clears at its first completed refresh. A missing artifact clears once a loadable one is installed on the search path, with no restart. A time zone mismatch clears once the server reports the zone that line was opened in, so align the servers' `timezone` setting; otherwise serve such tenants from a separate process. A table that does not compile is compiled again at every refresh, and clears once its schema, or the artifact answering for it, changes so that it compiles. The health probes stay green through all four — `/livez` and `/readyz` do not read the type layer — so watch for the ingest `503`s, for `wavehouse_sse_rows_withheld_total{reason="unavailable"}`, and for three `ERROR` log lines, each carrying the cause: `chtypes cannot serve this tenant` (a missing artifact or a time zone mismatch, logged at every refresh), `chtypes could not compile table schema` (one table, logged at every refresh) and `ingest type layer unavailable` (any of the four, logged for the ingest requests it refuses, at most once a minute per tenant and table). See [API → Ingest error responses](/api#post-v1ingesttabletable--ingest-data) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). **Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse never fetches an artifact itself: a line with no installed artifact makes the tenants on it unavailable at their schema refresh, and boot is refused only when no artifact is installed at all, or when an explicit `clickhouse.chtypes_registry` does not exist or cannot be read. One exception: the SDK honors `CHTYPES_AUTOFETCH=1` from the environment over WaveHouse's setting, and would then download a missing line (hundreds of MB) inside a schema refresh — leave it unset. diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index 0187607d..f9545fef 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -29,7 +29,7 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: scripts/fetch-chtypes.sh # wraps: go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --frozen --lock chtypes.lock 26.8 ``` -It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–300 MB — expect the first run to take a minute or two. Without it the API process refuses to boot (`make dev`, `make test-e2e`, and the app that `make test-integration` and `make ci` start), and the unit tests that need the engine skip; set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (CI does) to make a missing artifact fail those tests instead. A `503` on ingest, with row-filtered stream rows withheld, is what a process that did find an artifact answers for a ClickHouse line the artifact does not cover. +It lands in the default local cache (`~/.cache/chtypes/artifacts/abi6/-`, one directory per SDK ABI revision) and is 160–300 MB — expect the first run to take a minute or two. Without it the API process refuses to boot (`make dev`, `make test-e2e`, and the app that `make test-integration` and `make ci` start), and the unit tests that need the engine skip; set `WAVEHOUSE_TEST_REQUIRE_CHTYPES=1` (CI does) to make a missing artifact fail those tests instead. A process that did find an artifact answers ingest with `503`, and withholds row-filtered stream rows, for a ClickHouse line the artifact does not cover; that cause and the others are listed in [Deployment → chtypes artifacts](/deployment#chtypes-artifacts). ### Auto-installed by `make tools` diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index 7e8c49e9..1e9cbab2 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -41,7 +41,7 @@ The SDK **never throws** for anything the server returns — all API errors come | 502 | `clickhouse.misconfigured` | No | ClickHouse refused WaveHouse's own credentials or database, a read ran under a `readonly=1` profile and was refused as a write, or the route to it is wrong (a redirect, or a `4xx` other than `408`/`413`/`429`, with no exception code) — an operator fix | | 502 | `clickhouse.response_too_large` | No | A response over the 64 MiB cap, on any query path (structured query, pipe or `wh.sql`) | | 503 | `clickhouse.unavailable` | Yes | ClickHouse is down, unreachable or overloaded; `Retry-After: 5`, honored between attempts | -| 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, a schema not discovered yet, a tenant on no ClickHouse pool, a tenant whose ClickHouse line the type layer cannot serve (`ingest validation is unavailable`, `Retry-After: 5`), a dedupe store that cannot answer (`dedupe store unavailable`, `Retry-After: 5`), a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`), or a record whose dedupe id another request is still publishing (`a request with the same dedupe id is in flight`, `Retry-After`: the server's dedupe lease, 30 s by default). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on those last two causes waits that long; a stream re-dials on its own jittered backoff instead | +| 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, a schema not discovered yet, a tenant on no ClickHouse pool, a tenant or table the type layer cannot serve (`ingest validation is unavailable`, `Retry-After: 5`; [causes](/deployment#chtypes-artifacts)), a dedupe store that cannot answer (`dedupe store unavailable`, `Retry-After: 5`), a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`), or a record whose dedupe id another request is still publishing (`a request with the same dedupe id is in flight`, `Retry-After`: the server's dedupe lease, 30 s by default). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on those last two causes waits that long; a stream re-dials on its own jittered backoff instead | | 0 | `NETWORK_ERROR` | Yes | Network failure (retried with exponential backoff) | | 0 | `ABORTED` | No | Request canceled via `AbortSignal` | | 0 | `SSE_CONNECT_ERROR` | No | Stream could not be started (e.g. a non-absolute `baseURL`) | diff --git a/internal/api/ingest.go b/internal/api/ingest.go index 33c4f1c4..1b34d029 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -81,9 +81,9 @@ type IngestHandler struct { window int // noticeMu guards noticeLast, the last time each rate-limited notice was - // logged. A tenant the type layer cannot serve, or a policy whose injected - // literal will not compile, is a standing condition: one line per request - // would bury the rest of the log under it. + // logged. A tenant or table the type layer cannot serve, or a policy whose + // injected literal will not compile, is a standing condition: one line per + // request would bury the rest of the log under it. noticeMu sync.Mutex noticeLast map[string]time.Time } @@ -318,10 +318,10 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { return } - // A tenant the type layer cannot judge is refused before the body is read, - // like a schema not discovered yet: nothing in the body can change the - // answer, so there is no reason to buffer up to 16 MiB of it. The handle is - // not held across the read — a slow upload must not hold off a rebind. + // A tenant or table the type layer cannot judge is refused before the body + // is read, like a schema not discovered yet: nothing in the body can change + // the answer, so there is no reason to buffer up to 16 MiB of it. The handle + // is not held across the read — a slow upload must not hold off a rebind. if abort := h.typesReady(ctx, store.Tenant(), table); abort != nil { writeAbort(w, abort) return @@ -775,11 +775,12 @@ func (h *IngestHandler) typesFailed(ctx context.Context, id tenant.ID, table str } // unavailableAbort is the 503 for a type layer that cannot judge the tenant's -// table: a missing artifact for its ClickHouse line, a server zone this process -// cannot serve, a table that did not compile, or a tenant not bound yet. The -// body stays generic — the cause can name server paths and other tenants' -// zones, and is the operator's to read in the log. Retry-After is the schema -// hint's: the tenant's next schema refresh is what binds it again. +// table: a tenant not bound yet, a missing artifact for its ClickHouse line, a +// server zone this process cannot serve, or a table that did not compile (or, +// a wiring fault, no type layer at all). The body stays generic — the cause can +// name server paths and other tenants' zones, and is the operator's to read in +// the log. Retry-After is the schema hint's: the tenant's next schema refresh +// is what binds it again, and what compiles a failed table again. func unavailableAbort() *requestAbort { return &requestAbort{ Status: http.StatusServiceUnavailable, @@ -789,8 +790,9 @@ func unavailableAbort() *requestAbort { } // logUnavailable emits at most one line per tenant and table per minute. A -// missing artifact or a timezone mismatch persists until an operator acts, so -// the per-request line says nothing the first one didn't. +// missing artifact, a timezone mismatch or a table that does not compile +// persists until an operator acts, so the per-request line says nothing the +// first one didn't. func (h *IngestHandler) logUnavailable(ctx context.Context, id tenant.ID, table, cause string) { if h.noticeDue("unavailable:" + id.String() + "/" + table) { slog.ErrorContext(ctx, "ingest type layer unavailable", "tenant", id, "table", table, "cause", cause) diff --git a/internal/app/wire.go b/internal/app/wire.go index f50c3850..c72ec763 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -394,9 +394,9 @@ func (a *App) readConns(s *settings.Store) int { // the search path refuses boot: an API process could judge nothing. Opening // reads manifests only; a ClickHouse line's library is opened by the first // tenant bound to it, from that tenant's discovery (wireDiscovery), and a -// tenant whose line or zone this process cannot serve is unavailable on its -// own. Released after schema discovery, whose loops bind it, and so after -// the HTTP drain. +// tenant or table this process cannot serve is unavailable on its own +// (typelayer.Unavailable lists why). Released after schema discovery, whose +// loops bind it, and so after the HTTP drain. func (a *App) wireTypes() error { eng, err := typelayer.NewEngine(typelayer.Config{RegistryDir: a.cfg.ClickHouse.ChtypesRegistry}) if err != nil { diff --git a/internal/stream/metrics.go b/internal/stream/metrics.go index 68d1c1b6..017b8f60 100644 --- a/internal/stream/metrics.go +++ b/internal/stream/metrics.go @@ -92,13 +92,14 @@ func (m *Metrics) FrameDropped(kind string) { // tell "no matching rows" from "a misconfigured filter withholding everything", // and by reason (a Reason* constant, a closed set) so the cases that are a FAULT // rather than a filter verdict are visible on their own: `unavailable` means the -// type layer has no compiled schema for the tenant's table, so every -// row-filtered subscriber is dark until it does; `drift` means events arrive -// under a column list the table's current generation cannot read; `error` means -// ClickHouse raised evaluating the predicate over the row, or the type layer -// refused to parse the event at all; and `decline` means no verdict was reached -// — a filter that does not compile, a row that does not parse, or a filter on a -// column the inserting role did not write. Only `filter` is a policy decision. +// type layer has no compiled schema for the tenant's table (ReasonUnavailable +// lists why), so every row-filtered subscriber is dark until it does; `drift` +// means events arrive under a column list the table's current generation cannot +// read; `error` means ClickHouse raised evaluating the predicate over the row, or +// the type layer refused to parse the event at all; and `decline` means no +// verdict was reached — a filter that does not compile, a row that does not +// parse, or a filter on a column the inserting role did not write. Only `filter` +// is a policy decision. func (m *Metrics) RowWithheld(table, role, reason string) { if m == nil { return diff --git a/internal/stream/roweval.go b/internal/stream/roweval.go index 85fd6939..47757a88 100644 --- a/internal/stream/roweval.go +++ b/internal/stream/roweval.go @@ -39,9 +39,11 @@ const ( // like every answer that is not a definite true. ReasonDecline = typelayer.ReasonDecline // ReasonUnavailable: no compiled schema can answer for this tenant's table - // — the tenant is not bound yet, its server line has no artifact, a compile - // refusal, or no engine wired at all. Nothing about the row; every - // row-filtered subscriber of the table is affected until it is fixed. + // — the tenant is not bound yet, its server line has no artifact, its server + // zone differs from the one this process opened that line with, the table's + // schema did not compile, or no engine is wired at all. Nothing about the + // row; every row-filtered subscriber of the table is affected until it is + // fixed. ReasonUnavailable = "unavailable" // ReasonDrift: the event's column list names a column the table's current // generation does not export (dropped or renamed since), or names one twice, @@ -190,7 +192,7 @@ func (e *withheldError) Error() string { return e.err.Error() } func (e *withheldError) Unwrap() error { return e.err } // classifyPrepare names WHY no view could be prepared. The causes are -// operationally different — a missing artifact or an unbound tenant is an +// operationally different — an unavailable table (see ReasonUnavailable) is an // estate problem, drift is a schema change in flight — and they are // indistinguishable in the metric without the label. func classifyPrepare(err error) error { diff --git a/internal/typelayer/errors.go b/internal/typelayer/errors.go index 2285638d..d89ce935 100644 --- a/internal/typelayer/errors.go +++ b/internal/typelayer/errors.go @@ -12,9 +12,10 @@ import ( // and to "withhold" on the stream, so it must stay distinguishable from a // ClickHouse rejection. // -// It is always about one tenant: a missing artifact for the tenant's server -// line, a server zone this process cannot adopt, or a table that did not -// compile leaves every other tenant answering. +// It is always about one tenant, or one of its tables: a tenant not bound +// yet, a missing artifact for the tenant's server line, a server zone this +// process cannot adopt, or a table that did not compile (that table alone) +// leaves every other tenant answering. type Unavailable struct { Tenant tenant.ID // Table is "" when the cause covers the tenant's every table. diff --git a/internal/typelayer/typelayer.go b/internal/typelayer/typelayer.go index 92440f7b..ebdd39b9 100644 --- a/internal/typelayer/typelayer.go +++ b/internal/typelayer/typelayer.go @@ -70,10 +70,11 @@ type Config struct { // that tenant's successful refreshes (Bind), and the tenant's teardown drops // it (Forget). // -// Tenants are independent: a tenant whose server line has no artifact, whose -// server zone this process cannot adopt, or whose table does not compile is -// Unavailable on its own, and every other tenant keeps answering. Two tenants -// on the same server and database still compile separate handles. +// Tenants are independent: a tenant that is not bound yet, has no artifact for +// its server line, or reports a server zone this process cannot adopt is +// Unavailable on its own, a table that does not compile is Unavailable alone, +// and every other tenant keeps answering. Two tenants on the same server and +// database still compile separate handles. type Engine struct { reg *chtypes.Registry @@ -187,12 +188,13 @@ func (e *Engine) Table(id tenant.ID, name string) (*Table, error) { } // Bind resolves the library for the tenant's serverVersion and (re)compiles a -// handle per table, keeping every table whose columns and library are +// handle per table, keeping every compiled table whose columns and library are // unchanged. It is called synchronously from discovery's refresh hook, so it // must never be fatal: a failure is recorded as a cause — per table for a // compile refusal, tenant-wide for a missing artifact or a zone mismatch — -// and surfaces as *Unavailable from Table. A table the tenant no longer has -// is closed; a tenant whose line resolves again is answering again. +// and surfaces as *Unavailable from Table. A table whose compile was refused +// is compiled again at every Bind; a table the tenant no longer has is closed; +// a tenant whose line resolves again is answering again. // // serverTZ is the server's default zone name as ClickHouse reports it; "" // means UTC, chtypes' own default, never the host's zone. From e07e0ce39d8a6501edf64060981c4839a09b9f06 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 13:06:20 -0400 Subject: [PATCH 65/70] fix(policy): resolve row-filter predicates in column order resolvePredicates ranged over the filter map, so a filter over k columns rendered in up to k! orders across Evaluate calls. Each order compiled and cached its own chtypes filter on the stream (one compile per order, k times the cache entries) and split /v1/query's cache and singleflight keys. Iterate the columns sorted, as insertShape already does for checks. Co-Authored-By: Claude Opus 5.5 --- internal/policy/policy.go | 7 ++++++- internal/policy/policy_test.go | 29 +++++++++++++++++++++++++++++ 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/internal/policy/policy.go b/internal/policy/policy.go index 66836de7..bdbcd805 100644 --- a/internal/policy/policy.go +++ b/internal/policy/policy.go @@ -2,7 +2,9 @@ package policy import ( "fmt" + "maps" "regexp" + "slices" "strings" "github.com/Wave-RF/WaveHouse/internal/chsql" @@ -380,7 +382,10 @@ func resolvePredicates(filters map[string]Filter, claims map[string]any) []Predi } return Predicate{Column: col, Op: op, Values: []string{v}} } - for col, f := range filters { + // Sorted, so the same filter always renders the same text: the stream's + // compiled-filter cache and the query cache both key on it. + for _, col := range slices.Sorted(maps.Keys(filters)) { + f := filters[col] if f.Eq != nil { preds = append(preds, scalar(col, "=", *f.Eq)) } diff --git a/internal/policy/policy_test.go b/internal/policy/policy_test.go index 4eb1233f..74df2ab0 100644 --- a/internal/policy/policy_test.go +++ b/internal/policy/policy_test.go @@ -1688,6 +1688,35 @@ func TestPredicates_IsTheSameResolutionAsTheWhereClause(t *testing.T) { assert.Equal(t, []Predicate{{Column: "tenant_id", Op: "=", Values: []string{"acme"}}}, preds) } +// TestPredicates_OrderIsStableAcrossEvaluations: a filter over several +// columns resolves to the same predicate order (and so the same rendered text) +// on every Evaluate, so neither the stream's filter cache nor the query cache +// splits one filter into one entry per map iteration order. +func TestPredicates_OrderIsStableAcrossEvaluations(t *testing.T) { + t.Parallel() + p := &Policy{Tables: map[string]TablePolicy{ + "t": {"r": {Select: &SelectPermissions{Filter: map[string]Filter{ + "c": {Eq: new("{{ jwt.c }}")}, + "a": {Eq: new("{{ jwt.a }}")}, + "b": {Eq: new("{{ jwt.b }}")}, + "d": {Eq: new("{{ jwt.d }}")}, + }}}}, + }} + claims := map[string]any{"a": "1", "b": "2", "c": "3", "d": "4"} + first := Evaluate(p, "r", "t", "select", claims) + wantPreds, ok := first.Predicates() + require.True(t, ok) + assert.Equal(t, []string{"a", "b", "c", "d"}, []string{wantPreds[0].Column, wantPreds[1].Column, wantPreds[2].Column, wantPreds[3].Column}) + wantSQL, _ := first.Select.WhereSQL(nil) + for range 200 { + perms := Evaluate(p, "r", "t", "select", claims) + preds, _ := perms.Predicates() + require.Equal(t, wantPreds, preds) + sql, _ := perms.Select.WhereSQL(nil) + require.Equal(t, wantSQL, sql) + } +} + // TestPredicates_FailsClosedWhereNoRowMayBeAdmitted: a denied grant and an // INSERT-resolved grant refuse the row question rather than answer "no // predicates"; a nil receiver (no policy) and a resolved, unfiltered read side From 53d874ebe8b4e9d8e07c2d6a19f66042f357e769 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 13:08:21 -0400 Subject: [PATCH 66/70] refactor(api): inject the ClickHouse HTTP reader instead of a package global Co-Authored-By: Claude Sonnet 5.5 --- internal/api/clickhouse_http.go | 12 ++++++++---- internal/api/clickhouse_http_test.go | 17 +++++++++++++++++ internal/api/pipes.go | 6 +++++- internal/api/structured_query.go | 6 +++++- internal/app/wire.go | 3 +++ 5 files changed, 38 insertions(+), 6 deletions(-) diff --git a/internal/api/clickhouse_http.go b/internal/api/clickhouse_http.go index d9fbf605..ee7a695e 100644 --- a/internal/api/clickhouse_http.go +++ b/internal/api/clickhouse_http.go @@ -144,10 +144,14 @@ type readerPool struct { conns int } -// sharedCHReader serves both cached read handlers, so a tenant's reads share -// one connection cap whichever route they come in by, as they shared its -// native pool. -var sharedCHReader = newCHReader(readerHTTPClient) +// CHReader is the read client the structured-query and pipes handlers run +// their ClickHouse reads through. Give one to both (SetReader) so a tenant's +// reads share one connection cap whichever route they come in by, as they +// shared its native pool; a handler built without one has a private reader. +type CHReader = chReader + +// NewCHReader returns a reader over net/http with a client per pool. +func NewCHReader() *CHReader { return newCHReader(readerHTTPClient) } // readerHTTPClient is the read paths' client: net/http's default transport // with the target's TLS config and at most conns connections to the server, diff --git a/internal/api/clickhouse_http_test.go b/internal/api/clickhouse_http_test.go index 8c6e0575..eb69c1c2 100644 --- a/internal/api/clickhouse_http_test.go +++ b/internal/api/clickhouse_http_test.go @@ -556,3 +556,20 @@ func TestCheckRequestSize(t *testing.T) { func chExceptionBody(code int32, msg string) string { return fmt.Sprintf("Code: %d. DB::Exception: %s. (version 26.6.3.62 (official build))\n", code, msg) } + +// TestCHReader_HandlersShareOnlyWhenGiven: a handler built alone has a private +// reader, SetReader makes two handlers share one, and no package-level reader +// exists to share by accident. +func TestCHReader_HandlersShareOnlyWhenGiven(t *testing.T) { + sq := NewStructuredQueryHandler(nil, nil, nil, nil, nil, nil, nil) + ph := NewPipesHandler(nil, nil, nil, nil, nil) + if sq.ch == nil || ph.ch == nil || sq.ch == ph.ch { + t.Fatalf("unwired handlers must each own a private reader: %p %p", sq.ch, ph.ch) + } + r := NewCHReader() + sq.SetReader(r) + ph.SetReader(r) + if sq.ch != r || ph.ch != r { + t.Fatal("SetReader did not install the shared reader") + } +} diff --git a/internal/api/pipes.go b/internal/api/pipes.go index 34da7fce..e33ee4ac 100644 --- a/internal/api/pipes.go +++ b/internal/api/pipes.go @@ -52,9 +52,13 @@ type PipesHandler struct { } func NewPipesHandler(source func(*settings.Store) pipes.Source, policySource PolicySource, target func(*settings.Store) chconn.Target, c cache.Cache, queryTimeout func(*settings.Store) time.Duration) *PipesHandler { - return &PipesHandler{Source: source, PolicySource: policySource, Target: target, Cache: c, ch: sharedCHReader, queryTimeout: queryTimeout} + return &PipesHandler{Source: source, PolicySource: policySource, Target: target, Cache: c, ch: NewCHReader(), queryTimeout: queryTimeout} } +// SetReader replaces the handler's private reader with a shared one. Call it +// before serving. +func (h *PipesHandler) SetReader(r *CHReader) { h.ch = r } + // List returns all named queries of the ?tenant= (admin endpoint). func (h *PipesHandler) List(w http.ResponseWriter, r *http.Request) { store, ok := opsStore(w, r, h.Tenants) diff --git a/internal/api/structured_query.go b/internal/api/structured_query.go index 211b31b8..f6234530 100644 --- a/internal/api/structured_query.go +++ b/internal/api/structured_query.go @@ -65,7 +65,7 @@ func NewStructuredQueryHandler( ) *StructuredQueryHandler { return &StructuredQueryHandler{ Target: target, - ch: sharedCHReader, + ch: NewCHReader(), Cache: c, Registry: registry, PolicySource: policyStore, @@ -75,6 +75,10 @@ func NewStructuredQueryHandler( } } +// SetReader replaces the handler's private reader with a shared one. Call it +// before serving. +func (h *StructuredQueryHandler) SetReader(r *CHReader) { h.ch = r } + func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) { store, ok := requestStore(w, r) if !ok { diff --git a/internal/app/wire.go b/internal/app/wire.go index c72ec763..162fb3a5 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -1051,8 +1051,11 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { pipesHandler := api.NewPipesHandler(func(s *settings.Store) pipes.Source { return s }, (*settings.Store).Policy, a.chTargetFor, a.cache, queryTimeout) pipesHandler.Tenants = a.tenants pipesHandler.MaxConns = a.readConns + chReader := api.NewCHReader() + pipesHandler.SetReader(chReader) structuredQueryHandler := api.NewStructuredQueryHandler(a.chTargetFor, a.cache, a.registryFor, (*settings.Store).Policy, (*settings.Store).TimestampBucketSeconds, queryTimeout, (*settings.Store).DefaultMaxRows) structuredQueryHandler.MaxConns = a.readConns + structuredQueryHandler.SetReader(chReader) schemaHandler := api.NewSchemaHandler(a.registryFor) schemaHandler.Tenants = a.tenants From 63d7f455bb60e8de65b44924048e163f9366b443 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 13:16:51 -0400 Subject: [PATCH 67/70] test(app): run the package's tests in parallel where they share no globals internal/app took 8.7s alone under -race -cover (14-15s inside the full unit run, against its 15s timeout), almost all of it one test after another waiting on I/O and timers. Every test that touches no process-wide state now calls t.Parallel; the ones that read the default logger, set the environment, turn on Prometheus or signal the process stay serial, which keeps them apart from the parallel ones. The moved-tenant retry test waited on the discovery loop's shipped 2s backoff with full jitter, anywhere from a few milliseconds to several seconds. The loop's bounds are now fields set to the same 2s and 60s, which that test shortens to milliseconds. Alone under -race -cover the package now takes 4.2-4.7s, and 4.6s at GOMAXPROCS=2. No test removed; coverage is unchanged or higher. Co-Authored-By: Claude Opus 5.5 --- internal/app/app_test.go | 77 +++++++++++++++++++++++++++++++++--- internal/app/discoveries.go | 13 ++++-- internal/app/mq_nats_test.go | 3 ++ internal/app/roles_test.go | 7 ++++ internal/app/types_test.go | 4 ++ 5 files changed, 95 insertions(+), 9 deletions(-) diff --git a/internal/app/app_test.go b/internal/app/app_test.go index dfa06250..d17b8bad 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -43,11 +43,14 @@ import ( "github.com/Wave-RF/WaveHouse/internal/typelayer/typelayertest" ) -// None of these tests run in parallel: New installs a process-wide default -// logger and, with Prometheus on, the global OTel providers. Every app boots -// against a ClickHouse address that is guaranteed closed, so the boot-time -// schema discovery fails fast and deterministically (the degraded path) no -// matter what is listening on the developer's :9000. +// A test that touches process-wide state runs serially, which keeps it apart +// from every parallel test: one that reads the default logger +// (logtest.Capture, bootLogged), sets the environment, turns on Prometheus +// (New then installs the global OTel providers), or signals the process. Any +// other test may run in parallel. Every app boots against a ClickHouse +// address that is guaranteed closed, so the boot-time schema discovery fails +// fast and deterministically (the degraded path) no matter what is listening +// on the developer's :9000. // closedAddr returns a 127.0.0.1 address nothing listens on: bind an // ephemeral port, then release it. @@ -115,7 +118,8 @@ func testConfig(t *testing.T, settingsDir string) *config.Config { } // guardGlobals restores the process-wide state New may replace: the default -// logger, and the OTel providers when Prometheus/OTLP is on. +// logger, and the OTel providers when Prometheus/OTLP is on. Parallel tests +// call it too, only to silence their boots: none of them reads what it saved. func guardGlobals(t *testing.T) { t.Helper() savedLogger := slog.Default() @@ -161,6 +165,7 @@ func get(t *testing.T, h http.Handler, path string) *httptest.ResponseRecorder { } func TestNew_DegradedBootServesDiagnostics(t *testing.T) { + t.Parallel() cfg := testConfig(t, writeSettings(t, nil)) a := newApp(t, cfg, Options{Build: BuildInfo{Version: "1.2.3", GitCommit: "abc", BuildTime: "now"}}) @@ -185,6 +190,7 @@ func TestNew_DegradedBootServesDiagnostics(t *testing.T) { // answer — boot is degraded without ClickHouse — so it proves the tenant // resolved. func TestNew_TenantHeaderResolvesAgainstTheRegistry(t *testing.T) { + t.Parallel() a := newApp(t, testConfig(t, writeSettings(t, nil)), Options{}) tests := []struct { @@ -199,6 +205,7 @@ func TestNew_TenantHeaderResolvesAgainstTheRegistry(t *testing.T) { } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { + t.Parallel() req := httptest.NewRequestWithContext(t.Context(), http.MethodGet, tt.path, nil) if tt.header != "" { req.Header.Set(tenant.Header, tt.header) @@ -223,6 +230,7 @@ func TestAsyncGetters_RegistryMiss(t *testing.T) { } func TestNew_DedupeFollowsSettings(t *testing.T) { + t.Parallel() tests := []struct { name string enabled bool @@ -232,6 +240,7 @@ func TestNew_DedupeFollowsSettings(t *testing.T) { } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { + t.Parallel() dir := writeSettings(t, map[string]any{"dedupe": map[string]any{ "enabled": tt.enabled, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}, }}) @@ -255,6 +264,7 @@ func rewriteSettings(t *testing.T, dir string, patch map[string]any) { } func TestReload_DrivesTheRegisteredHooks(t *testing.T) { + t.Parallel() // Hooks are registered in New and fired by the reload triggers Run // starts; a direct Reload stands in for any of the three triggers and // pins that the relocated hooks still follow the adopted document. @@ -356,6 +366,7 @@ func TestNew_NestedDirectory(t *testing.T) { // admin token cannot — over a nested directory the ops routes reach every // tenant, so the operator key alone opens them. func TestNew_NestedOperatorReloadsOneTenant(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "broken": invalidQuery}) cfg := testConfig(t, root) cfg.Auth.OperatorKey = "unit-test-operator-key" @@ -430,6 +441,7 @@ func TestNew_NestedWithoutAnOperatorKeyWarnsTheOpsTreeIsClosed(t *testing.T) { // request, so a lost 0 folder is felt at once on the routes that read tenant // 0's list. func TestReload_NestedHooksFollowEachTenant(t *testing.T) { + t.Parallel() dedupeOn := map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}} grown := map[string]any{"dedupe": dedupeOn, "mq": map[string]any{"max_bytes_gb": 2}} root := writeNestedSettings(t, map[string]map[string]any{ @@ -503,6 +515,7 @@ func TestReload_NestedHooksFollowEachTenant(t *testing.T) { // reopened over the same seen ids when the folder is back. The instance is // open while some tenant's store is, and Close releases it. func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { + t.Parallel() dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}}} root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn, "globex": nil, "broken": invalidQuery}) cfg := testConfig(t, root) @@ -565,6 +578,7 @@ func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { // is reached only by a Config built by hand; it must refuse boot, not wire // nothing. func TestNew_RefusesALayerWithoutABackend(t *testing.T) { + t.Parallel() for _, tc := range []struct { key string unset func(*config.Config) @@ -575,6 +589,7 @@ func TestNew_RefusesALayerWithoutABackend(t *testing.T) { {"coord.backend", func(c *config.Config) { c.Coord.Backend = "" }}, } { t.Run(tc.key, func(t *testing.T) { + t.Parallel() guardGlobals(t) cfg := testConfig(t, writeSettings(t, nil)) tc.unset(cfg) @@ -590,6 +605,7 @@ func TestNew_RefusesALayerWithoutABackend(t *testing.T) { // instance — their ingest answers 503 until a reload or a restart opens it — // while the process, and every tenant with dedupe off, carries on. func TestNew_DedupeOpenFailure(t *testing.T) { + t.Parallel() dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}}} // A regular file where the instance's directory should be is what Pebble // refuses to open. @@ -598,6 +614,7 @@ func TestNew_DedupeOpenFailure(t *testing.T) { require.NoError(t, os.WriteFile(filepath.Join(dataDir, "pebble"), nil, 0o600)) } t.Run("flat refuses boot", func(t *testing.T) { + t.Parallel() guardGlobals(t) cfg := testConfig(t, writeSettings(t, dedupeOn)) block(t, cfg.DataDir) @@ -605,6 +622,7 @@ func TestNew_DedupeOpenFailure(t *testing.T) { require.ErrorContains(t, err, "dedupe open") }) t.Run("nested fails closed", func(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn, "globex": dedupeOn, "initech": nil}) cfg := testConfig(t, root) block(t, cfg.DataDir) @@ -628,6 +646,7 @@ func TestNew_DedupeOpenFailure(t *testing.T) { // force the failure, as TestNew_DedupeOpenFailure does Pebble's. The failed // open clears it, so the next publish opens the queue: each one tries again. func TestNew_QueueOpenFailure(t *testing.T) { + t.Parallel() block := func(t *testing.T, dataDir, stream string) { t.Helper() p := filepath.Join(dataDir, "nats", "jetstream", "$G", "streams", stream) @@ -635,6 +654,7 @@ func TestNew_QueueOpenFailure(t *testing.T) { require.NoError(t, os.WriteFile(p, nil, 0o600)) } t.Run("flat refuses boot", func(t *testing.T) { + t.Parallel() guardGlobals(t) cfg := testConfig(t, writeSettings(t, nil)) block(t, cfg.DataDir, "DLQ_0") @@ -642,6 +662,7 @@ func TestNew_QueueOpenFailure(t *testing.T) { require.ErrorContains(t, err, "mq open") }) t.Run("nested costs the tenant alone", func(t *testing.T) { + t.Parallel() // globex, not acme: opened first, acme's streams keep the streams // directory occupied through globex's failed open, which the server // would otherwise remove on a goroutine of its own while the next @@ -660,6 +681,7 @@ func TestNew_QueueOpenFailure(t *testing.T) { // Boot opens each served tenant's queue under New's context, as New's doc // says: a stop signaled during boot is not held up by one open per tenant. func TestNew_QueueSetupHonorsTheBootContext(t *testing.T) { + t.Parallel() guardGlobals(t) ctx, cancel := context.WithCancel(t.Context()) cancel() @@ -707,6 +729,7 @@ func TestSharedTables_InvalidatesTheTenantsSharingTheTables(t *testing.T) { // the wiring orphans its cache as it comes back; a tenant that stayed is // never touched, and a reload that changes nothing bumps nobody. func TestReload_ReadmittedTenantCacheIsOrphaned(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) a := newApp(t, testConfig(t, root), Options{}) // The hooks read a.cache at reload time: a recording cache from here on. @@ -758,6 +781,7 @@ func (p *pruneRecorder) last() map[tenant.ID]bool { // Every reload prunes the cache's version index down to the tenants served, // so a tenant rejected or removed stops holding it (#262). func TestReload_PrunesCacheIndexToServedTenants(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) a := newApp(t, testConfig(t, root), Options{}) _, ok := a.cache.(pruner) @@ -795,6 +819,7 @@ func redisTestConfig(t *testing.T, settingsDir, addr string) *config.Config { // reached does not refuse boot: the cache starts bypassed, and the reload // hook that prunes an in-process index leaves it alone. func TestNew_RedisCacheBootsBypassedWhenUnreachable(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) a := newApp(t, redisTestConfig(t, root, closedAddr(t)), Options{}) _, ok := a.cache.(*cache.RedisCache) @@ -813,6 +838,7 @@ func TestNew_RedisCacheBootsBypassedWhenUnreachable(t *testing.T) { // A TLS file that went missing between validation and wiring refuses boot, // naming the key. func TestNew_RedisCacheRefusesAnUnreadableTLSFile(t *testing.T) { + t.Parallel() guardGlobals(t) cfg := redisTestConfig(t, writeSettings(t, nil), closedAddr(t)) cfg.Cache.Redis.TLS = config.CacheRedisTLS{Enabled: true, CAFile: filepath.Join(t.TempDir(), "gone.pem")} @@ -885,6 +911,7 @@ func keepalive(interval, buckets int) map[string]any { // longer ones are inside of (#597 tracks honoring each tenant's own). A flat // directory's single tenant gets exactly its own pair. func TestShortestKeepalive(t *testing.T) { + t.Parallel() open := func(t *testing.T, dir string) *settings.Registry { t.Helper() guardGlobals(t) @@ -894,12 +921,14 @@ func TestShortestKeepalive(t *testing.T) { } t.Run("flat directory", func(t *testing.T) { + t.Parallel() period, buckets := shortestKeepalive(open(t, writeSettings(t, keepalive(45, 5)))) assert.Equal(t, 45*time.Second, period) assert.Equal(t, 5, buckets) }) t.Run("nested directory", func(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": keepalive(30, 3), "globex": keepalive(10, 2), "initech": keepalive(10, 7)}) tenants := open(t, root) period, buckets := shortestKeepalive(tenants) @@ -916,6 +945,7 @@ func TestShortestKeepalive(t *testing.T) { }) t.Run("no tenant served falls back to the wheel's defaults", func(t *testing.T) { + t.Parallel() period, buckets := shortestKeepalive(open(t, writeNestedSettings(t, map[string]map[string]any{"acme": invalidQuery}))) assert.Zero(t, period) assert.Zero(t, buckets) @@ -933,6 +963,7 @@ func gapWindow(minutes int) map[string]any { // (mq.Purger.PurgeAcked). A flat directory's single tenant gets exactly its // own window. func TestGapWindows(t *testing.T) { + t.Parallel() open := func(t *testing.T, dir string) *settings.Registry { t.Helper() guardGlobals(t) @@ -942,10 +973,12 @@ func TestGapWindows(t *testing.T) { } t.Run("flat directory", func(t *testing.T) { + t.Parallel() assert.Equal(t, map[tenant.ID]time.Duration{tenant.Default: 45 * time.Minute}, gapWindows(open(t, writeSettings(t, gapWindow(45))))) }) t.Run("nested directory", func(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": gapWindow(15), "globex": gapWindow(60), "initech": gapWindow(30)}) tenants := open(t, root) assert.Equal(t, map[tenant.ID]time.Duration{"acme": 15 * time.Minute, "globex": 60 * time.Minute, "initech": 30 * time.Minute}, gapWindows(tenants)) @@ -962,6 +995,7 @@ func TestGapWindows(t *testing.T) { }) t.Run("a folder rejected since boot keeps everything", func(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": invalidQuery}) tenants := open(t, root) assert.Equal(t, map[tenant.ID]time.Duration{"acme": keepEverything}, gapWindows(tenants)) @@ -976,6 +1010,7 @@ func TestGapWindows(t *testing.T) { // A finding about a nested directory itself — a loose file beside the tenant // folders — refuses boot, like an invalid flat directory. func TestNew_NestedLooseFileRefusesBoot(t *testing.T) { + t.Parallel() guardGlobals(t) root := writeNestedSettings(t, map[string]map[string]any{"acme": nil}) require.NoError(t, os.WriteFile(filepath.Join(root, "notes.txt"), []byte("scratch"), 0o600)) @@ -986,6 +1021,7 @@ func TestNew_NestedLooseFileRefusesBoot(t *testing.T) { } func TestNew_RefusesInvalidSettingsDirectory(t *testing.T) { + t.Parallel() guardGlobals(t) cfg := testConfig(t, t.TempDir()) // empty: every required file is missing a, err := newForTest(t.Context(), t, Options{Config: cfg}) @@ -1045,6 +1081,7 @@ func analystPipe(t *testing.T, dir string) { // place) once the dedupe store is already open, and a second New on the same // data_dir must find the Pebble lock released. func TestNew_LateBootFailureReleasesEverything(t *testing.T) { + t.Parallel() guardGlobals(t) dir := writeSettings(t, map[string]any{"dedupe": map[string]any{ "enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}, @@ -1070,6 +1107,7 @@ func TestNew_LateBootFailureReleasesEverything(t *testing.T) { // than evaluated under the default_role, and the process serves everything // else. func TestNew_UnreachableJWKSBootsFailClosed(t *testing.T) { + t.Parallel() dir := writeSettings(t, authPatch("http://"+closedAddr(t)+"/jwks.json")) analystPipe(t, dir) started := time.Now() @@ -1100,6 +1138,7 @@ func TestNew_UnreachableJWKSBootsFailClosed(t *testing.T) { // is refused under globex's header, and pointing acme's folder at another // provider and reloading it swaps acme's verifier alone. func TestNew_VerifierPerTenant(t *testing.T) { + t.Parallel() acme, acmeKey, _ := jwksServer(t, "acme-1") globex, globexKey, globexFetches := jwksServer(t, "globex-1") root := writeNestedSettings(t, map[string]map[string]any{ @@ -1248,6 +1287,7 @@ func TestRun_ServesUntilCancelled(t *testing.T) { } func TestRun_SweeperRunsUnderItsLease(t *testing.T) { + t.Parallel() var lc net.ListenConfig ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") require.NoError(t, err) @@ -1302,6 +1342,7 @@ func TestRun_PrometheusSidecar(t *testing.T) { } func TestRun_ListenFailureStopsEverything(t *testing.T) { + t.Parallel() // Hold the wildcard address the server binds (":port"), not loopback: a // process already on the port holds the same address, and macOS allows a // wildcard bind while only 127.0.0.1:port is held (Linux refuses both), @@ -1349,6 +1390,7 @@ func (c dyingConsumer) Consume(func(*mq.Message), int) (func(), <-chan error, er } func TestRun_DeadIngestWorkerStopsEverything(t *testing.T) { + t.Parallel() a := newApp(t, testConfig(t, writeSettings(t, nil)), Options{}) // The worker takes its consumer from a.mq when Run starts it. a.mq = dyingConsumerBroker{Broker: a.mq, reason: fmt.Errorf("%w: consumer deleted", mq.ErrDeliveryEnded)} @@ -1364,6 +1406,7 @@ func TestRun_DeadIngestWorkerStopsEverything(t *testing.T) { } func TestClose_AbandonsAStuckCloseAtTheDeadline(t *testing.T) { + t.Parallel() release := make(chan struct{}) t.Cleanup(func() { close(release) }) a := &App{} @@ -1436,6 +1479,7 @@ func endsCleanly(t *testing.T, stream io.Reader) { // reconnect then meets the tenant's 503 or 404. A flat directory never stops // serving tenant 0, so a reload it rejects leaves the stream open. func TestRun_StopEndsOpenStreams(t *testing.T) { + t.Parallel() start := func(t *testing.T, settingsDir string) (a *App, baseURL string, stop func() error) { t.Helper() var lc net.ListenConfig @@ -1454,6 +1498,7 @@ func TestRun_StopEndsOpenStreams(t *testing.T) { } t.Run("the stop", func(t *testing.T) { + t.Parallel() _, baseURL, stop := start(t, writeSettings(t, nil)) resp := openStream(t, baseURL, "") started := time.Now() @@ -1463,6 +1508,7 @@ func TestRun_StopEndsOpenStreams(t *testing.T) { }) t.Run("a reload that stops serving the tenant", func(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) a, baseURL, stop := start(t, root) defer func() { assert.NoError(t, stop()) }() @@ -1494,6 +1540,7 @@ func TestRun_StopEndsOpenStreams(t *testing.T) { }) t.Run("a reload the flat directory rejects", func(t *testing.T) { + t.Parallel() dir := writeSettings(t, nil) a, baseURL, stop := start(t, dir) resp := openStream(t, baseURL, "") @@ -1523,6 +1570,7 @@ func poolSettings(addr string, open int) map[string]any { // config and the settings pool must fit under it, so an impossible pair // refuses to boot naming both numbers (#530). func TestNew_RefusesAPoolAboveTheCeiling(t *testing.T) { + t.Parallel() guardGlobals(t) cfg := testConfig(t, writeSettings(t, poolSettings(closedAddr(t), 10))) cfg.ClickHouse.MaxTotalConns = 4 @@ -1577,6 +1625,7 @@ func chSettings(addr, user string, open int) map[string]any { // its own and leaves the other on the very same Manager, resized to its own // ask — the worked example of the story. func TestNew_NestedPoolsFollowEachTenantsTuple(t *testing.T) { + t.Parallel() shared, other := closedAddr(t), closedAddr(t) root := writeNestedSettings(t, map[string]map[string]any{ "acme": chSettings(shared, "default", 10), @@ -1609,6 +1658,7 @@ func TestNew_NestedPoolsFollowEachTenantsTuple(t *testing.T) { // A nested directory's pools must fit the ceiling together: boot is refused // naming the sum and the ceiling, like a flat directory's one pool. func TestNew_NestedRefusesPoolsAboveTheCeiling(t *testing.T) { + t.Parallel() guardGlobals(t) root := writeNestedSettings(t, map[string]map[string]any{ "acme": chSettings(closedAddr(t), "default", 10), @@ -1675,6 +1725,7 @@ func TestReload_CeilingRefusesAThirdTupleThenOpensIt(t *testing.T) { // fresh pool, registry and verifier, and an id it sent before the removal is // still a duplicate. Its open streams end too: TestRun_StopEndsOpenStreams. func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { + t.Parallel() jwks, _, fetches := jwksServer(t, "acme-1") acmeSettings := authPatch(jwks.URL) acmeSettings["dedupe"] = map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "retention": "0", "tables": map[string]any{}} @@ -1755,6 +1806,7 @@ func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { // another tenant's discovery does afterwards; /readyz then pings every open // pool and names each one that does not answer. func TestNew_NestedProbesFollowTheFirstTenantToLoad(t *testing.T) { + t.Parallel() root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) a := newApp(t, testConfig(t, root), Options{}) @@ -1791,6 +1843,7 @@ func TestNew_NestedProbesFollowTheFirstTenantToLoad(t *testing.T) { // Retry-After, not a 404: in a flat directory during the degraded boot, and // in a nested one per tenant. func TestNew_SchemaNotLoadedIs503(t *testing.T) { + t.Parallel() ingest := func(t *testing.T, a *App, id string) *httptest.ResponseRecorder { t.Helper() req := httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/v1/ingest?table=clicks", strings.NewReader(`{"page": "/"}`)) @@ -1803,6 +1856,7 @@ func TestNew_SchemaNotLoadedIs503(t *testing.T) { return rec } t.Run("flat, degraded boot", func(t *testing.T) { + t.Parallel() a := newApp(t, testConfig(t, writeSettings(t, nil)), Options{}) rec := ingest(t, a, "") assert.Equal(t, http.StatusServiceUnavailable, rec.Code, "body: %s", rec.Body.String()) @@ -1810,6 +1864,7 @@ func TestNew_SchemaNotLoadedIs503(t *testing.T) { assert.Contains(t, rec.Body.String(), "schema not loaded yet") }) t.Run("nested, per tenant", func(t *testing.T) { + t.Parallel() a := newApp(t, testConfig(t, writeNestedSettings(t, map[string]map[string]any{"acme": nil})), Options{}) rec := ingest(t, a, "acme") assert.Equal(t, http.StatusServiceUnavailable, rec.Code, "body: %s", rec.Body.String()) @@ -1890,6 +1945,7 @@ func databaseSettings(addr, database string) map[string]any { // schema.refresh_interval (60 seconds here). The tenant beside it, and a // reload that moves nobody, keep the registry and the loop they had. func TestReload_MovedTenantDiscoversTheNewDatabase(t *testing.T) { + t.Parallel() addr := closedAddr(t) root := writeNestedSettings(t, map[string]map[string]any{ "acme": databaseSettings(addr, "default"), @@ -1931,6 +1987,7 @@ func TestReload_MovedTenantDiscoversTheNewDatabase(t *testing.T) { // A flat directory's tenant 0 moves the same way. func TestReload_MovedTenantDiscoversTheNewDatabase_Flat(t *testing.T) { + t.Parallel() addr := closedAddr(t) dir := writeSettings(t, databaseSettings(addr, "default")) a := newApp(t, testConfig(t, dir), Options{}) @@ -1958,6 +2015,11 @@ func TestReload_MovedTenantFailedDiscoveryIsRetried(t *testing.T) { root := writeNestedSettings(t, map[string]map[string]any{"acme": databaseSettings(addr, "default")}) a := newApp(t, testConfig(t, root), Options{}) fake := newFakeClickHouse(t, a, map[string][]string{"default": {"events"}}) + // Retried within milliseconds, not the shipped 2s doubling to 60s with + // jitter, so the retry that finds the database does not wait on chance. + a.discoveries.mu.Lock() + a.discoveries.backoff, a.discoveries.maxBackoff = 10*time.Millisecond, 20*time.Millisecond + a.discoveries.mu.Unlock() logs := logtest.Capture(t, slog.LevelWarn) // No fake names moved_db: its discovery dials the closed port. @@ -2007,6 +2069,7 @@ func (c *registryAtInvalidation) InvalidateTenant(ctx context.Context, id tenant // pool, and a request arriving meanwhile must not find the previous // database's schema. func TestReload_MovedTenantRegistryIsDroppedBeforeTheCacheInvalidation(t *testing.T) { + t.Parallel() addr := closedAddr(t) root := writeNestedSettings(t, map[string]map[string]any{"acme": databaseSettings(addr, "default")}) a := newApp(t, testConfig(t, root), Options{}) @@ -2045,6 +2108,7 @@ func (c *stuckConn) Query(context.Context, string, ...any) (driver.Rows, error) // the reload returns: Close waits for it with the loops of the served // tenants, and names its tenant when the release budget ends first. func TestClose_WaitsForALoopAReloadStopped(t *testing.T) { + t.Parallel() conn := &stuckConn{entered: make(chan struct{}), release: make(chan struct{})} // Released on every way out, so a failed assertion leaves no loop stuck. release := sync.OnceFunc(func() { close(conn.release) }) @@ -2075,6 +2139,7 @@ func TestClose_WaitsForALoopAReloadStopped(t *testing.T) { // Close stops every tenant's discovery loop within the release budget, and // the pools after them. func TestClose_StopsTheDiscoveryLoops(t *testing.T) { + t.Parallel() a := newApp(t, testConfig(t, writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil})), Options{}) loops := *a.discoveries.cur.Load() require.Len(t, loops, 2) diff --git a/internal/app/discoveries.go b/internal/app/discoveries.go index b54305d9..f8bb549e 100644 --- a/internal/app/discoveries.go +++ b/internal/app/discoveries.go @@ -40,6 +40,9 @@ type discoveries struct { // onRetire, when set, is told each registry a reload retires, under mu // and the reload lock: what it does must not wait on I/O. onRetire func(tenant.ID, *discovery.SchemaRegistry) + // backoff and maxBackoff bound a loop's retry before its first success + // (RetryRefresh's jittered backoff), read under mu as each loop starts. + backoff, maxBackoff time.Duration mu sync.Mutex // serializes reconcile, drop, adopt and close cur atomic.Pointer[map[tenant.ID]*tenantDiscovery] @@ -58,7 +61,10 @@ type tenantDiscovery struct { } func newDiscoveries(ctx context.Context, build func(tenant.ID, *settings.Store) *discovery.SchemaRegistry, onAttempt func(tenant.ID, error), onLoaded func(tenant.ID), onRetire func(tenant.ID, *discovery.SchemaRegistry)) *discoveries { - d := &discoveries{ctx: ctx, build: build, onAttempt: onAttempt, onLoaded: onLoaded, onRetire: onRetire} + d := &discoveries{ + ctx: ctx, build: build, onAttempt: onAttempt, onLoaded: onLoaded, onRetire: onRetire, + backoff: 2 * time.Second, maxBackoff: 60 * time.Second, + } d.cur.Store(&map[tenant.ID]*tenantDiscovery{}) return d } @@ -141,14 +147,15 @@ func (d *discoveries) adopt(id tenant.ID, reg *discovery.SchemaRegistry) { } // start runs reg's loop: the boot retry until the first success, skipped -// for a registry already loaded, then the periodic refresh. +// for a registry already loaded, then the periodic refresh. Under mu. func (d *discoveries) start(id tenant.ID, reg *discovery.SchemaRegistry) *tenantDiscovery { ctx, cancel := context.WithCancel(d.ctx) td := &tenantDiscovery{id: id, registry: reg, cancel: cancel, done: make(chan struct{})} + backoff, maxBackoff := d.backoff, d.maxBackoff go func() { defer close(td.done) if !reg.Loaded() { - err := reg.RetryRefresh(ctx, 2*time.Second, 60*time.Second, func(err error) { d.onAttempt(id, err) }) + err := reg.RetryRefresh(ctx, backoff, maxBackoff, func(err error) { d.onAttempt(id, err) }) if err != nil { // ctx cancelled before success — the process is stopping, or // the tenant is no longer served. diff --git a/internal/app/mq_nats_test.go b/internal/app/mq_nats_test.go index dd26e785..8ef07fea 100644 --- a/internal/app/mq_nats_test.go +++ b/internal/app/mq_nats_test.go @@ -38,6 +38,7 @@ func natsConfig(t *testing.T, url string) *config.Config { // consumes the operator's durable; the operator deleting it ends the worker, // and with it Run, naming the component. func TestNew_NATSBackend(t *testing.T) { + t.Parallel() srv := natstest.Start(t) cfg := natsConfig(t, srv.URL()) cfg.Roles = []config.Role{config.RoleAPI, config.RoleIngest} @@ -99,6 +100,7 @@ func TestNew_NATSWiresNoSweeper(t *testing.T) { //nolint:paralleltest // capture // A cluster never reached within topology_wait refuses boot as unavailable. func TestNew_NATSUnreachable(t *testing.T) { + t.Parallel() guardGlobals(t) cfg := natsConfig(t, "nats://"+closedAddr(t)) cfg.MQ.NATS.TopologyWait = time.Millisecond @@ -109,6 +111,7 @@ func TestNew_NATSUnreachable(t *testing.T) { // The operator's topology missing a piece refuses boot with the finding. func TestNew_NATSTopologyMissing(t *testing.T) { + t.Parallel() srv := natstest.Start(t) require.NoError(t, srv.Operator.JetStream().DeleteStream(t.Context(), "WH_DLQ")) guardGlobals(t) diff --git a/internal/app/roles_test.go b/internal/app/roles_test.go index e6536f1e..f5e1ad84 100644 --- a/internal/app/roles_test.go +++ b/internal/app/roles_test.go @@ -28,6 +28,7 @@ import ( // process's. New does not validate, so the embedded MQ stands in for the // shared one a split needs (config.Validate refuses it outside tests). func TestNew_RolesChooseTheComponents(t *testing.T) { + t.Parallel() for _, tc := range []struct { name string roles []config.Role @@ -60,6 +61,7 @@ func TestNew_RolesChooseTheComponents(t *testing.T) { }}, } { t.Run(tc.name, func(t *testing.T) { + t.Parallel() cfg := testConfig(t, writeSettings(t, nil)) cfg.Roles = tc.roles a := newApp(t, cfg, Options{}) @@ -107,6 +109,7 @@ func TestNew_OnlyTheAPIRoleNeedsTheArtifact(t *testing.T) { } func TestNew_RefusesAConfigWithoutRoles(t *testing.T) { + t.Parallel() guardGlobals(t) cfg := testConfig(t, writeSettings(t, nil)) cfg.Roles = nil @@ -128,6 +131,7 @@ func hs256(t *testing.T, secret, role string) string { // token verifier runs there, so even an admin token the API would admit is // refused. Every tenant route, and the rest of /v1/ops, is not there. func TestNew_OpsOnlyRouter(t *testing.T) { + t.Parallel() dir := writeSettings(t, nil) require.NoError(t, os.WriteFile(filepath.Join(dir, settings.FileRoles), []byte(`{"roles": ["admin"]}`), 0o600)) require.NoError(t, os.WriteFile(filepath.Join(dir, settings.FilePolicies), []byte(`{"admin_role": "admin", "tables": {}}`), 0o600)) @@ -187,6 +191,7 @@ func TestNew_OpsOnlyRouter(t *testing.T) { // ClickHouse pool answers (here none can), a sweeper-only one once booted. // Liveness never waits on schema discovery, which only the API runs. func TestNew_OpsOnlyReadiness(t *testing.T) { + t.Parallel() cfg := testConfig(t, writeSettings(t, nil)) cfg.Roles = []config.Role{config.RoleIngest} a := newApp(t, cfg, Options{}) @@ -199,6 +204,7 @@ func TestNew_OpsOnlyReadiness(t *testing.T) { // A reload that moves the tenant repoints the pool of a process without the // api role, which has no schema registry to start over. func TestReload_OpsOnlyMovedTenant(t *testing.T) { + t.Parallel() addr := closedAddr(t) dir := writeSettings(t, databaseSettings(addr, "default")) cfg := testConfig(t, dir) @@ -225,6 +231,7 @@ func TestNew_OpsOnlyPrometheusInline(t *testing.T) { // A sweeper-only process serves its listener and runs the sweeper under the // lease, as the all-roles one does. func TestRun_SweeperOnlyProcess(t *testing.T) { + t.Parallel() var lc net.ListenConfig ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") require.NoError(t, err) diff --git a/internal/app/types_test.go b/internal/app/types_test.go index 0ee7bcff..d8013ed9 100644 --- a/internal/app/types_test.go +++ b/internal/app/types_test.go @@ -65,6 +65,7 @@ func registryOf(id tenant.ID, table string) *discovery.SchemaRegistry { // retired; its retirement forgets the tenant, and a refresh that outlives it // binds nothing. func TestTypeBindings_TheOwnerBindsUntilRetired(t *testing.T) { + t.Parallel() rec := &recordingBinder{} b := newTypeBindings(rec) reg := registryOf("acme", "a") @@ -85,6 +86,7 @@ func TestTypeBindings_TheOwnerBindsUntilRetired(t *testing.T) { // tables never stay bound over the new one's. A late detach of the retired // registry leaves the successor's binding alone. func TestTypeBindings_RetiredMidBindNeverOutlivesTheSuccessor(t *testing.T) { + t.Parallel() rec := &recordingBinder{} b := newTypeBindings(rec) old, successor := registryOf("acme", "old_db"), registryOf("acme", "new_db") @@ -120,6 +122,7 @@ func TestTypeBindings_RetiredMidBindNeverOutlivesTheSuccessor(t *testing.T) { // Tenants bind independently: one tenant's retirement forgets only its own. func TestTypeBindings_TenantsAreIndependent(t *testing.T) { + t.Parallel() rec := &recordingBinder{} b := newTypeBindings(rec) acme, globex := registryOf("acme", "a"), registryOf("globex", "g") @@ -136,6 +139,7 @@ func TestTypeBindings_TenantsAreIndependent(t *testing.T) { // drop retires (a tenant moved to another database) and the one a reconcile // retires (a tenant no longer served). func TestDiscoveries_ReportEveryRetiredRegistry(t *testing.T) { + t.Parallel() type retired struct { id tenant.ID reg *discovery.SchemaRegistry From 26015744f351d51b8e49f80d09a5968818b44a17 Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 13:46:59 -0400 Subject: [PATCH 68/70] fix(ingest): decline a body chtypes answers short, never report it short A UUID value shorter than 36 characters makes ClickHouse's UUID reader consume a fixed 36-byte window past it, and its error recovery resumes at the end of the line that window ended on. The records the window reaches get no verdict while the batch outcome stays Accepted, so a JSON array, NDJSON, CSV or TSV body lost them silently: `total` was chtypes' short count and nothing said records were missing. Measured on 26.8.15.10; a server INSERT with input_format_allow_errors_ratio loses the same rows. The type layer now counts the records itself (typelayer/records.go): the JSON array's element count from the handler, exact, or a floor over NDJSON top-level objects and CSV/TSV lines (blank lines included, quoted and escaped newlines not) less a declared or auto-detected header. When chtypes answers fewer (for an array, any other number) the whole body is declined: one 422 per counted record, nothing published, total is the count, Batch.Miscount set and logged once. A body chtypes read whole never trips it. A JSON array with an empty element is now a 400, since it frames as a blank line that is no record. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- clients/ts/src/table.ts | 8 +- docs/src/content/docs/api.md | 24 +- docs/src/content/docs/architecture.md | 7 +- docs/src/content/docs/sdk/queries.md | 2 +- internal/api/ingest.go | 35 ++- internal/api/ingest_count_test.go | 162 +++++++++++ internal/api/ingest_framing.go | 30 ++- internal/api/ingest_framing_test.go | 5 + internal/typelayer/ingest.go | 122 +++++++-- internal/typelayer/records.go | 370 ++++++++++++++++++++++++++ internal/typelayer/records_test.go | 265 ++++++++++++++++++ 12 files changed, 985 insertions(+), 47 deletions(-) create mode 100644 internal/api/ingest_count_test.go create mode 100644 internal/typelayer/records.go create mode 100644 internal/typelayer/records_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index 919154af..14873cb0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -63,7 +63,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `project_name`, `git.ignore_tags` and `builds:`, and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes` with `--notes-start-tag` set to the previous `v*` tag, `git describe --tags --abbrev=0 --match 'v*' "${tag}^"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. -- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser in one call per request, as-is except that a JSON array is re-framed in place to one element per line so one bad element cannot lose the rest, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable (`33` for a record cut off mid-object), `117` unknown field, otherwise the type's own code (`72`, `41`, `38`, `69`, `376`, `467`, `691`, `675` …); an out-of-range integer wraps exactly as ClickHouse's own `INSERT` does (`256` → `0` in a `UInt8`) — so a malformed record is no longer the whole-request `400 {"error":"invalid json"}` (a JSON array whose brackets do not balance, or with content after its `]`, is still a whole-request `400 {"error":"invalid json: …"}`); a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; and **timestamp values on the wire — published rows, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date` and `Date32` keep their `"2026-06-21"` form on the stream, and on `/v1/query` and pipes they lose the `T00:00:00Z` the native path added — see the structured-query entry) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant not bound yet, one whose ClickHouse line has no installed artifact, or one whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line) is refused on its own, and so is a table whose schema chtypes could not compile (that table alone) — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable` — while every other tenant keeps working. Each cause is checked again at every schema refresh and clears at the first one after it is fixed ([Deployment → chtypes artifacts](https://wavehouse.dev/deployment#chtypes-artifacts)). Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A dedupe id is read from the exported row: a `null` cell or an empty string is missing (so is an omitted `String` id with no `DEFAULT`), and a numeric id column cannot tell an omitted `0` from a supplied one. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser in one call per request, as-is except that a JSON array is re-framed in place to one element per line so one bad element cannot lose the rest, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable (`33` for a record cut off mid-object), `117` unknown field, otherwise the type's own code (`72`, `41`, `38`, `69`, `376`, `467`, `691`, `675` …); an out-of-range integer wraps exactly as ClickHouse's own `INSERT` does (`256` → `0` in a `UInt8`) — so a malformed record is no longer the whole-request `400 {"error":"invalid json"}` (a JSON array whose brackets do not balance, or with content after its `]`, is still a whole-request `400 {"error":"invalid json: …"}`, and so is one with an empty element — a leading, doubled or trailing comma); a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; a body chtypes answers fewer records for than WaveHouse counts in it — a `UUID` shorter than 36 characters makes ClickHouse's reader consume the records after it, which then get no verdict while the batch is still accepted — is declined whole, every counted record `422` and nothing published, rather than reported as a smaller batch; and **timestamp values on the wire — published rows, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date` and `Date32` keep their `"2026-06-21"` form on the stream, and on `/v1/query` and pipes they lose the `T00:00:00Z` the native path added — see the structured-query entry) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant not bound yet, one whose ClickHouse line has no installed artifact, or one whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line) is refused on its own, and so is a table whose schema chtypes could not compile (that table alone) — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable` — while every other tenant keeps working. Each cause is checked again at every schema refresh and clears at the first one after it is fixed ([Deployment → chtypes artifacts](https://wavehouse.dev/deployment#chtypes-artifacts)). Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A dedupe id is read from the exported row: a `null` cell or an empty string is missing (so is an omitted `String` id with no `DEFAULT`), and a numeric id column cannot tell an omitted `0` from a supplied one. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. - **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, instead of walking a decoded record's keys. A column the role may not write stays in that schema as `MATERIALIZED` of its default, so naming it is refused while expressions that read it keep working and the stored row holds the table's default. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a non-`Nullable` checked column behaves exactly like omitting it (on a `Nullable` one it stores `NULL`, which fails the check). An `_eq` check also covers a column the role may not otherwise write: a record that omits it is filled with the required value and published, one that supplies exactly that value is accepted (**0.1.0 answered `403 column "x" not allowed for insert` to the correct value**), and any other value is `403 check failed for column "x"`. A role whose schema cannot be compiled this way, or that may write no column of the table, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After` (the cause is logged once a minute), rather than a `503` that a retry could not fix; a policy `check` on an `EPHEMERAL` column is still `403`. diff --git a/clients/ts/src/table.ts b/clients/ts/src/table.ts index 37eb8da8..f0dfb7e0 100644 --- a/clients/ts/src/table.ts +++ b/clients/ts/src/table.ts @@ -98,9 +98,11 @@ export class TableRef> { * * A single object is sent as a JSON `POST /v1/ingest`. An **array** is * serialized to NDJSON (one record per line) and sent as a single - * `application/x-ndjson` request: a bad record no longer fails or hides the - * rest of the batch — per-record outcomes come back in the result - * (`failed` / `results`), and `ok` is true only when every record succeeded. + * `application/x-ndjson` request: a bad record does not fail the rest of + * the batch — per-record outcomes come back in the result (`failed` / + * `results`), and `ok` is true only when every record succeeded. A body the + * server's parser cannot read record by record is declined whole: every + * record fails with `validation engine declined: …` and none is inserted. * * The array path sends one request regardless of size, so it is bound by the * server's 16 MiB request-body cap (an over-cap array is a `413` with nothing diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 06f587db..078fa2f5 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -236,7 +236,7 @@ Validates a body of records against the ClickHouse schema for `{table}` and publ | `text/csv; header=present`, `text/tab-separated-values; header=present` | a header line naming the columns, in any order — see [Header formats](#header-formats-headerpresent) | | anything else, or none | `415`, listing the accepted types | -The two JSON families are one format to ClickHouse; the declaration decides only how the body frames its records. The single thing the body still chooses is *arity within `application/json`*: the first non-whitespace byte picks an array (`[`) or a single object. Under a single-object body only the first object is answered — concatenated objects after it are parsed but neither published nor reported, a `200` for one record (one cut off mid-record can still turn that answer into a `422` decline); declare NDJSON for anything line-framed ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). The reverse now works when every element is valid: a JSON array declared `application/x-ndjson` ingests every element. It is not re-framed, though, so one bad element in a single-line array makes chtypes decline the whole body (`422` for every record); declare `application/json` to get per-record answers. +The two JSON families are one format to ClickHouse; the declaration decides only how the body frames its records. The single thing the body still chooses is *arity within `application/json`*: the first non-whitespace byte picks an array (`[`) or a single object. Under a single-object body only the first object is answered — concatenated objects after it are parsed but neither published nor reported, a `200` for one record (one cut off mid-record, or a short `UUID` in the first that takes the next with it, can still turn that answer into a `422` decline); declare NDJSON for anything line-framed ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). The reverse now works when every element is valid: a JSON array declared `application/x-ndjson` ingests every element. It is not re-framed, though, so one bad element in a single-line array makes chtypes decline the whole body (`422` for every record); declare `application/json` to get per-record answers. :::note[What counts as a valid declaration] The header is parsed with Go's `mime.ParseMediaType` (RFC 9110 §8.3) and the **media type** decides the format, so no malformed *parameter* costs the request — `application/json; charset`, `application/json;;`, a value left mid-quote, a name repeated with different values all read as `application/json`. The one parameter that also decides a format is `header`, on `text/csv` and `text/tab-separated-values` only: `present` selects the header format, `absent` the strictly positional one, no `header` at all ClickHouse's default reading, and any other value is a `415`. A line whose parameters did not parse and that mentions `header` is a `415` as well, because guessing at it could ingest a declared header line as data or drop a data row as a header. Two more things are refused. A malformed parameter on a line that **also contains a comma** is a `415`, because the comma may be a second declaration joined on and the error cannot tell that from a comma inside data ([#563](https://github.com/Wave-RF/WaveHouse/issues/563)) — so `application/json; profile="a,b"` is fine and `application/json; profile="a,b"; charset` is not. And `Content-Type` is a **singleton** field (§5.3 forbids repeating it), so repeated header *lines* are accepted only when they agree, while a comma-joined value is refused outright: §8.3 warns that picking a member of the resulting pseudo-list is itself an interoperability and security hazard. @@ -258,7 +258,7 @@ The policy engine authorizes mutations by inspecting the columns being written. - An omitted column takes its `DEFAULT` expression — evaluated by ClickHouse, including a volatile one like `now()` — or the type's default where none is declared (`NULL` on a `Nullable` column), exactly as an `INSERT` naming fewer columns does. An explicit `null` does the same on a non-`Nullable` column (WaveHouse pins `input_format_null_as_default`); on a `Nullable` column it stores `NULL`. - A coercion ClickHouse would make it makes here (a numeric string into an `Int*`, JSON `true` into an `Int*` as `1`, an out-of-range integer wrapping; a quoted `"true"` into a `Bool` it refuses, code 467); anything it would refuse fails synchronously in the ingest response with its real code, rather than surfacing later in the DLQ. `Nullable()` and `LowCardinality()` wrappers are transparent. -WaveHouse decides only policy: whether the role may insert at all, and whether the record satisfies the role's [`check` clauses](/access-control#insert-checks) — evaluated by the same compiled-filter engine as row-level security, in the same parse that validates the record and against the row ClickHouse produced, so a check sees stored values rather than the payload's spelling. A record ClickHouse refuses reports that refusal, never a check result. A record chtypes did not answer for — as opposed to accepting or rejecting it — is **declined** (`422`), which is not a verdict on that record's data: it can be the shape, or a body the reader could not get through as a whole (below), in which case every record in the body is declined. +WaveHouse decides only policy: whether the role may insert at all, and whether the record satisfies the role's [`check` clauses](/access-control#insert-checks) — evaluated by the same compiled-filter engine as row-level security, in the same parse that validates the record and against the row ClickHouse produced, so a check sees stored values rather than the payload's spelling. A record ClickHouse refuses reports that refusal, never a check result. A record chtypes did not answer for — as opposed to accepting or rejecting it — is **declined** (`422`), which is not a verdict on that record's data: it can be the shape, or a body the reader could not get through record by record (below), in which case every record in the body is declined. **Error responses.** Rows marked **per-record** are reported in `results` on a batch body (the request itself stays `200`) and become the response status on a single-object body; every other row fails the whole request. @@ -270,6 +270,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26, 27 or 33) — or, cut right after a key's `:` or an opening quote, a `422` decline of every record in the body | | 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes, or only whitespace (only the first 512 bytes are looked at, so a body opening with 512 bytes of whitespace counts as empty too). A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | +| 400 | `{"error":"invalid json: empty element in the json array (a leading, doubled or trailing comma)"}` | A body declared `application/json` opening with `[` holds an empty element: `[,{…}]`, `[{…},,{…}]` or `[{…},]`. Its element count would not be the record count, so the whole request fails and nothing is published | | 400 | `{"error":"invalid json: content after the closing ']' of the json array"}` | A body declared `application/json` opening with `[` has something other than whitespace after its closing `]`. That tail is not a record of the array, so the whole request fails and nothing is published | | 400 | `{"error":"missing dedupe id field \"event_id\""}` | **Per-record.** Only with `dedupe.require_id: true`, when the row carries no value for the configured `id_field` (an absent column, a `null` cell or an empty string — the value an omitted `String` id column stores). With `require_id: false` (the default) the row is published un-deduped instead. Either way it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (the gate surfaces the token reason rather than silently falling back to `default_role`) | @@ -280,7 +281,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 404 | `{"error":"unknown table: ..."}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, application/ndjson, application/jsonl, application/jsonlines, text/csv, text/csv; header=present, text/csv; header=absent, text/tab-separated-values, text/tab-separated-values; header=present, text/tab-separated-values; header=absent"}` (declared variant: `Content-Type "text/plain": ingest requires one of …`; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | No `Content-Type`, an unsupported or unparseable one, a `header` value other than `present`/`absent`, a comma-bearing value that does not parse as a single media type, or repeated lines that disagree. Checked before the body is read | -| 422 | `{"error":"validation engine declined: "}` | **Per-record.** chtypes did not answer for the record. That can be the shape (the artifact declined it), or a body it could not read as a whole — a record cut off right after a key's `:` or an opening quote, or a single-line array declared NDJSON holding a bad element — in which case every record in the body, good ones included, gets this answer and nothing is published. A `check` clause that could not be evaluated lands here too (`validation engine declined: the insert check for column "x" could not be evaluated`) | +| 422 | `{"error":"validation engine declined: "}` | **Per-record.** chtypes did not answer for the record. That can be the shape (the artifact declined it), or a body it could not read record by record — a record cut off right after a key's `:` or an opening quote, a single-line array declared NDJSON holding a bad element, or a body it answered fewer records for than WaveHouse counts in it (`the body holds … records but chtypes answered …`; see [Batch Ingest](#batch-ingest)) — in which case every record in the body, good ones included, gets this answer and nothing is published. A `check` clause that could not be evaluated lands here too (`validation engine declined: the insert check for column "x" could not be evaluated`) | | 500 | `{"error":"validation failed"}` | The parse itself failed for a reason that is neither the record's fault nor an unavailable tenant or table (the `503` below); logged. Nothing was published | | 500 | `{"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` | The role's insert permissions do not compile against the table (for example, the role may write no column of it). It persists until the policy or the table changes, so it carries no `Retry-After` and must not be retried; the cause is in the server log | | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | @@ -332,7 +333,8 @@ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling Cl | too few fields | rejected, code **27** — ClickHouse's own message, e.g. `Cannot parse input: expected ',' before: …` | | too many fields | rejected, code **117** — `Expected end of line` | | a header line, no `header` parameter | ClickHouse detects it and **consumes** it as a header: `total` and every `index` count data rows only | -| a header line, `header=absent` | **not a header** — read as a data row, so it fails to parse wherever a column cannot read its own name, with that column type's own code (72 for a `Float64`, for example); the data rows after it still parse | +| a header line, `header=absent` | **not a header** — read as a data row, so it fails to parse wherever a column cannot read its own name, with that column type's own code (72 for a `Float64`, for example); the data rows after it still parse — except that a `UUID` column's reader can take the next row with the name, which declines the whole body ([Batch Ingest](#batch-ingest)) | +| a blank line | a record, not skipped: a one-column table stores the column's empty value, a wider one rejects it | The messages are ClickHouse's own and differ between ClickHouse lines; branch on the `exception_code`. An empty **TSV** field is the empty string, not a default: `\N` is TSV's `NULL`, which a non-`Nullable` column turns into its default, and a `DateTime64` cannot read `""`. @@ -367,9 +369,14 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ #### Batch Ingest -A JSON array, an NDJSON body, a CSV body or a TSV body ingests a batch in one request. Each record is validated, authorized, deduplicated and published independently, so **one malformed or rejected record never blocks the rest of the batch** — including inside a single-line (compact) JSON array declared `application/json`, which WaveHouse re-frames in place before handing it over. The exception is a body ClickHouse's reader cannot get through at all — a record cut off right after a key's `:` or an opening quote, or a single-line array declared NDJSON that holds a bad element: chtypes then declines the whole body, and every record it read answers `422`, with nothing published. ClickHouse's JSON reader also treats a newline as whitespace, so an NDJSON line cut off mid-value can absorb the line after it: the pair is answered as one rejected record, and `total` counts it once. An explicit empty array (`[]`) is a valid record-less batch (`200`, `total: 0`) for a role whose insert grant resolves; blank lines in an NDJSON body are skipped. (The SDK's `insert([...])` array helper uses the NDJSON form automatically; every form returns the same response.) +A JSON array, an NDJSON body, a CSV body or a TSV body ingests a batch in one request. Each record is validated, authorized, deduplicated and published independently, so **one malformed or rejected record never blocks the rest of the batch** — including inside a single-line (compact) JSON array declared `application/json`, which WaveHouse re-frames in place before handing it over. The exceptions are bodies ClickHouse's reader cannot get through record by record, and nothing in them is published: -The response counts records read, published, rejected and deduplicated, then lists per-record outcomes: each entry mirrors the single-object response (`ok` / `duplicate` / `error`) plus its 1-based `index`, and carries ClickHouse's numeric `exception_code` when the rejection was its parser's. `results` is truncated to the first 10,000 entries; the four counts stay authoritative. +- A record cut off right after a key's `:` or an opening quote, or a single-line array declared NDJSON that holds a bad element: chtypes declines the whole body, and every record answers `422`. +- A value whose reader consumes past its own record. Measured for a `UUID` shorter than 36 characters: its reader takes the next 36 bytes of the body with it, and the records those bytes reach get no verdict of their own (a ClickHouse server inserting the same body with `input_format_allow_errors_ratio` set loses the same records: its `INSERT` succeeds without them). WaveHouse counts the records in the body itself — a JSON array's elements, the top-level objects of an NDJSON body, the lines of a CSV or TSV body (a blank line included) less a header — and when chtypes answers fewer (for a JSON array, any other number) it declines the whole body instead of reporting a smaller batch: every counted record answers `422`, `total` is that count, and the message names both counts and the first record ClickHouse refused, which is where the records went missing. Outside a JSON array the count is a floor rather than exact: a malformed body can only lower it (an object cut off mid-value hides the ones after it, and a first CSV or TSV line WaveHouse cannot decode is taken for a header), never raise it, so a body chtypes read whole is never declined. + +ClickHouse's JSON reader also treats a newline as whitespace, so an NDJSON line cut off mid-value can absorb the line after it: the pair is answered as one rejected record, and `total` counts it once. An explicit empty array (`[]`) is a valid record-less batch (`200`, `total: 0`) for a role whose insert grant resolves; blank lines in an NDJSON body are skipped. (The SDK's `insert([...])` array helper uses the NDJSON form automatically; every form returns the same response.) + +The response counts the body's records (`total`) and those published, rejected and deduplicated, then lists per-record outcomes: each entry mirrors the single-object response (`ok` / `duplicate` / `error`) plus its 1-based `index`, and carries ClickHouse's numeric `exception_code` when the rejection was its parser's. `results` is truncated to the first 10,000 entries; the four counts stay authoritative. ```bash curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ @@ -393,7 +400,7 @@ curl -X POST "http://localhost:8080/v1/ingest?table=clicks" \ | Field | Meaning | | ----- | ------- | -| `total` | records read from the body | +| `total` | records in the body: the ones chtypes answered for, or, when the batch is declined whole because it answered short, the ones WaveHouse counted | | `succeeded` | records validated and published | | `failed` | records rejected — see `results` | | `duplicates` | records skipped by dedup (when enabled) | @@ -406,6 +413,7 @@ A `200` is returned whenever the body was read and the records were processed | 400 | `{"error":"empty body"}` (declared variants: `empty ndjson body`, `empty csv body`, `empty tsv body`, `empty csvwithnames body`, `empty tsvwithnames body`) | The body holds no bytes, or only whitespace (only the first 512 bytes are looked at, so a body opening with 512 bytes of whitespace counts as empty too). A `header=present` body holding only its header line is a valid record-less batch (`200`, `total: 0`) | | 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26, 27 or 33) — or, cut right after a key's `:` or an opening quote, a `422` decline of every record in the body | | 400 | `{"error":"invalid json: unterminated json array"}` | A body declared `application/json` opening with `[` whose brackets do not balance — truncated, or structurally broken. It cannot be salvaged per record, so the whole request fails | +| 400 | `{"error":"invalid json: empty element in the json array (a leading, doubled or trailing comma)"}` | A body declared `application/json` opening with `[` holds an empty element: `[,{…}]`, `[{…},,{…}]` or `[{…},]`. Its element count would not be the record count, so the whole request fails and nothing is published | | 400 | `{"error":"invalid json: content after the closing ']' of the json array"}` | A body declared `application/json` opening with `[` has something other than whitespace after its closing `]`. That tail is not a record of the array, so the whole request fails and nothing is published | | 400 | `{"error":"Unknown field found in format header: 'x' at position 1 …","code":"clickhouse.rejected","exception_code":117}` | A `header=present` body whose header names a column the table — or the role's writable set — does not have, or names one twice; see the [single-record table](#post-v1ingesttabletable--ingest-data). Nothing is published | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (same auth gate as the single-object path; surfaces the token reason) | @@ -426,7 +434,7 @@ A `200` is returned whenever the body was read and the records were processed | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] -A batch aborted partway — a `503` or `500` after some leading records were already published — re-publishes those leading records when the whole batch is retried. Records are published in windows of 256, in order: a dedupe failure drops the open window unpublished, so what an aborted batch published is the windows before it, plus, after a publish failure, the records of its window before the failing one. Failures decided before any record is processed are not in this class: a `413`, a `415`, the `400 invalid request body` of an upload cut off in transit, the unterminated-array `400` and the `400 invalid json: content after the closing ']' of the json array` all publish nothing and are safe to retry as-is (once split, for a `413`). Enable deduplication if duplicate suppression matters — the single-object path has the same at-least-once property, and the SDK retries both on `503`. +A batch aborted partway — a `503` or `500` after some leading records were already published — re-publishes those leading records when the whole batch is retried. Records are published in windows of 256, in order: a dedupe failure drops the open window unpublished, so what an aborted batch published is the windows before it, plus, after a publish failure, the records of its window before the failing one. Failures decided before any record is processed are not in this class: a `413`, a `415`, the `400 invalid request body` of an upload cut off in transit, the unterminated-array `400`, the empty-element `400` and the `400 invalid json: content after the closing ']' of the json array` all publish nothing and are safe to retry as-is (once split, for a `413`). Enable deduplication if duplicate suppression matters — the single-object path has the same at-least-once property, and the SDK retries both on `503`. ::: --- diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 623dd32c..6811feb3 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -85,7 +85,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`sql_classify.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. -- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (brackets that do not balance are a whole-request `400 invalid json: unterminated json array`, anything but whitespace after the closing `]` a whole-request `400 invalid json: content after the closing ']' of the json array`, and a `…WithNames` header naming an unknown or repeated column a whole-request `400 clickhouse.rejected` with code 117, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. A tenant or table the engine cannot answer for (the tenant not bound yet, no artifact for its ClickHouse line, a server time zone that differs from the one this process opened that line with, or a table schema the parser could not compile) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read, without affecting other tenants. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's duplicate window (two minutes on the embedded broker) — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). +- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (brackets that do not balance are a whole-request `400 invalid json: unterminated json array`, an empty element — a leading, doubled or trailing comma — a whole-request `400 invalid json: empty element in the json array …`, anything but whitespace after the closing `]` a whole-request `400 invalid json: content after the closing ']' of the json array`, and a `…WithNames` header naming an unknown or repeated column a whole-request `400 clickhouse.rejected` with code 117, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. Those verdicts account for every record or for none: the type layer counts the body's records itself (`typelayer/records.go`: the array's element count, passed in as `IngestOptions.Records`, or a floor over NDJSON objects and CSV/TSV lines less a header), because a value whose reader consumes past its own record — a `UUID` shorter than 36 characters — leaves the records after it with no verdict while the batch is still accepted, and when chtypes answers fewer (for an array, any other number) the whole body is declined (`Batch.Miscount`, one `422` per counted record, logged once) rather than reported short. A tenant or table the engine cannot answer for (the tenant not bound yet, no artifact for its ClickHouse line, a server time zone that differs from the one this process opened that line with, or a table schema the parser could not compile) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read, without affecting other tenants. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's duplicate window (two minutes on the embedded broker) — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` and `/` left unescaped via `output_format_json_escape_forward_slashes=0`, the same spellings the structured-query path and the SSE wire use, so a timestamp and a `/` read the same on every surface. - **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `output_format_json_named_tuples_as_objects=1`, `output_format_json_escape_forward_slashes=0`, `date_time_output_format=iso`, so a `/` is not escaped and a timestamp is RFC 3339 in UTC, both as on the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so `chconn.Classify` and `writeCHError` apply as on every other ClickHouse path, and a response past the 64 MiB cap is `clickhouse.response_too_large`. The read paths hold at most the tenant's pool size in HTTP connections — its `max_open_conns`, or the largest among the tenants sharing its connection tuple (address, database, user, password, TLS) — one cap per tuple, like the native pool; tenants on one server with a different database, user, password or TLS settings each have their own. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). @@ -171,7 +171,7 @@ Each tenant has its own table set inside that engine, bound from the tenant's ow - **`Engine.Bind`** runs synchronously from `discovery.SchemaRegistry`'s `OnRefresh` hook for one tenant, on every successful refresh: it resolves the artifact matching that tenant's server **minor** version — never a nearest-version fallback — and recompiles a handle per table whose column signature changed since the last bind or whose last compile failed, and every table when the tenant's library changes. Until its first bind a tenant is unavailable on its own (not bound yet). So is a tenant whose ClickHouse line has no installed artifact, and one whose server time zone differs from the zone this process already opened that line with: chtypes takes its time zone once per process, when a line is first opened, so one process serves one server time zone per ClickHouse line. Those two are logged as `chtypes cannot serve this tenant`, and a later bind of the same tenant that resolves its line (a refresh after the artifact is installed, or once the server reports that line's zone) clears them. A table whose schema chtypes could not compile is unavailable on its own, logged as `chtypes could not compile table schema`, and is compiled again at every bind until it compiles. In every case the cause is recorded (an ingest client sees only a generic `503`; the cause goes to the log), and every other tenant keeps working, as do the tenant's other tables when only one table failed to compile. [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) has the operator's view: how each cause recovers and what to watch. - **`Engine.RoleTable(tenant, table, shape)`** compiles the role's *own* schema — the columns it may insert, plus a `DEFAULT ''` on each `_eq` check column, including one the role may not otherwise write — and caches it per generation and shape. That is how column policy and auto-inject are answered without WaveHouse looking at a record: a denied column stays in the schema as `MATERIALIZED` of its default, so naming it is ClickHouse's code 117 while expressions that read it still compile, and an absent check column takes the claim as its default while a supplied value is accepted only if it equals the claim. A role-shape pool starts at one handle and grows to `min(GOMAXPROCS, 4)` when every handle is busy; a base table's grows to `min(GOMAXPROCS, 8)`. -- **`Table.IngestWith(format, opts, body, checks...)`** (`Ingest` is the same with no options) runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. +- **`Table.IngestWith(format, opts, body, checks...)`** (`Ingest` is the same with no options) runs the whole request body through ClickHouse's own reader in one call (`JSONEachRow`, `CSV`, `TSV`, `CSVWithNames` or `TSVWithNames`), with the parsing settings the worker's `INSERT` pins (`date_time_input_format=best_effort`, `input_format_null_as_default=1`) and unknown fields refused. It returns one verdict per input record — **accepted**, **rejected** with ClickHouse's real code and message, or **declined** (chtypes could not answer at all, a distinct condition never conflated with a rejection) — plus the accepted rows as `JSONCompactEachRow` bytes, exactly what ClickHouse's own writer produced: `DEFAULT`s evaluated, out-of-range integers wrapped, computed columns absent. When chtypes answers fewer records than the body holds — counted by `records.go`, exactly from the caller's `IngestOptions.Records` or as a floor over NDJSON objects and CSV/TSV lines less a detected or declared header — every counted record is declined instead and `Batch.Miscount` says why. - **Predicates** compile through chtypes with every bound value as a `{pN:String}` parameter, never interpolated — on an integer column wrapped in the same strict round-trip cast (`chsql.StrictInt`) the query builder emits, so a claim that does not fit the column matches nothing instead of wrapping. `IngestWith` judges an ingest `check` in the same parse that validates the body; `Table.ParseRow(columns, row)` / `Row.Visible` judge a subscriber's row filter over one parsed event, read under the column list the event carries (a column-restricted role's narrower list included), with compiled filters cached per handle (up to 4096 on each). Only a definite true admits; a predicate error, a policy column the table no longer has, a filter column the published row does not carry (`decline`), schema drift, or an unavailable engine all withhold (fail closed), counted in `wavehouse_sse_rows_withheld_total{table,role,reason}`. - A tenant or table that is unavailable answers ingest with `503` and `Retry-After: 5`, and the stream withholds every row of that tenant's tables (or of that one table) from a role that has a row `filter`, with reason `unavailable`. - A role whose schema cannot be compiled — or that may write no column of the table — is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After`, and the cause is logged once a minute; it is not the `503` an unavailable tenant or table gets, since retrying cannot fix a policy or a table. A shape that fails only because an injected check value cannot be its column's default (`'abc'` on a `UInt64`) is retried without the defaults and served: a record omitting that column then fails the check (`403` on an integer column, `422` on another), and only a shape that does not compile even without them gets the `500`. @@ -291,7 +291,8 @@ Client POST /v1/ingest?table={table} MATERIALIZED/ALIAS column, or a column this role may not write is 117 (an EPHEMERAL value is accepted only where the format names columns, the role may write it, a DEFAULT reads it and no computed column does; it feeds that DEFAULT, never stored or published); a record chtypes did not answer for is a distinct "declined" outcome - (422) — the shape, or a body it could not read as a whole, which + (422) — the shape, or a body it could not read record by record + (including one it answered fewer records for than the body holds), which declines every record in it → Evaluate the role's check clauses over the accepted rows with one compiled chtypes filter — false is 403 for that record, unevaluable is 422 diff --git a/docs/src/content/docs/sdk/queries.md b/docs/src/content/docs/sdk/queries.md index 0bc79b1c..7e079e91 100644 --- a/docs/src/content/docs/sdk/queries.md +++ b/docs/src/content/docs/sdk/queries.md @@ -26,7 +26,7 @@ const { data } = await clicks.fetch({ limit: 50, signal: controller.signal }); ### `.insert(data, opts?)` -Insert one row or many. A single object is sent as a JSON `POST /v1/ingest?table={table}`. An **array** is serialized to NDJSON (one record per line) and sent as a single `application/x-ndjson` request, so a bad record no longer fails or hides the rest of the batch — per-record outcomes come back in the result. +Insert one row or many. A single object is sent as a JSON `POST /v1/ingest?table={table}`. An **array** is serialized to NDJSON (one record per line) and sent as a single `application/x-ndjson` request, so a bad record does not fail the rest of the batch — per-record outcomes come back in the result. The exception is a body ClickHouse's reader cannot get through record by record (a short `UUID` that takes the records after it with it, for one): it is declined whole, every record a `422` and nothing inserted ([Batch Ingest](/api#batch-ingest)). ```ts // Single row → { ok: true } (or { ok: true, duplicate: true } when dedup skips it) diff --git a/internal/api/ingest.go b/internal/api/ingest.go index 1b34d029..93917e84 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -209,6 +209,11 @@ type ingestRun struct { // before chtypes has read it (an empty array is zero, anything else is at // least one), then len(batch.Rows) once it has answered. records int + // framed is a JSON array's element count — the one body whose framing + // gives WaveHouse an exact record count before chtypes reads it — and 0 + // for every other body. chtypes answering any other number declines the + // whole body (typelayer.IngestOptions.Records). + framed int // checkColumns names the check clauses a record's check answer came from, // for the rejection message. The filter is AND-joined over all of them, so // a false verdict does not say which one failed — with one clause it does. @@ -375,12 +380,18 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { } records = n } + framed := 0 + if format == FormatJSON && first == '[' { + framed = records + } // Otherwise a single-object body is one record, and a line-framed body has // at least the record its first byte starts. Concatenated objects after a // single object are neither answered nor published, as they always have // been (declare NDJSON to batch them, #561), but chtypes still parses // them, so one cut off mid-record can turn the answer into a decline. The - // real count is chtypes' own, taken once it has answered. + // real count is chtypes' own once it has answered — held to a floor the + // type layer counts in the body itself, so a short answer is a decline, + // never a short batch. guard := h.policyCheckGuard(ctx, table, role, schema, perms) shape, preds, checkColumns, abort := h.insertShape(ctx, table, role, schema, perms, guard) @@ -391,7 +402,7 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { run := &ingestRun{ store: store, table: table, scope: scope, now: now, - records: records, checkColumns: checkColumns, checkGuard: guard, + records: records, framed: framed, checkColumns: checkColumns, checkGuard: guard, } if records > 0 { if abort := h.judge(ctx, run, shape, format, body.Bytes(), preds); abort != nil { @@ -443,7 +454,9 @@ func (h *IngestHandler) judge(ctx context.Context, run *ingestRun, shape typelay if abort != nil { return abort } - batch, err := tbl.IngestWith(format.wire(), format.options(), body, preds...) + opts := format.options() + opts.Records = run.framed + batch, err := tbl.IngestWith(format.wire(), opts, body, preds...) run.wire = slices.Clone(tbl.WireColumns) tbl.Release() if err != nil { @@ -457,11 +470,19 @@ func (h *IngestHandler) judge(ctx context.Context, run *ingestRun, shape typelay slog.WarnContext(ctx, "ingest body refused by the parser", "error", r.Message, "exception_code", r.Code, "table", run.table) return &requestAbort{Status: http.StatusBadRequest, Message: r.Message, Code: codeCHRejected, ExceptionCode: r.Code} } + if m := batch.Miscount; m != nil { + // Every record is answered declined (422) and nothing is published; + // logged once here because the cause is the batch's, not any record's. + slog.ErrorContext(ctx, "chtypes answered a different number of records than the body holds; declining the whole batch", + "counted", m.Counted, "exact", m.Exact, "verdicts", m.Verdicts, + "table", run.table, "format", format.String()) + } run.batch = batch - // chtypes' per-record answer is the record count: a JSON array sent as - // NDJSON is however many elements its reader took, a blank line is nothing. - // When it gave no per-record detail the padded batch is still index-shaped, - // so the same rule keeps every later index in range. + // The type layer's answer is the record count: chtypes' own verdicts when + // they account for every record in the body, or one declined verdict per + // counted record when they do not or chtypes gave no per-record detail. + // A JSON array sent as NDJSON is however many elements its reader took, + // a blank NDJSON line is nothing. run.records = len(batch.Rows) return nil } diff --git a/internal/api/ingest_count_test.go b/internal/api/ingest_count_test.go new file mode 100644 index 00000000..b07e2cc2 --- /dev/null +++ b/internal/api/ingest_count_test.go @@ -0,0 +1,162 @@ +package api + +import ( + "net/http" + "net/http/httptest" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/testutil" +) + +const testUUID = "61f0c404-5cb3-11e7-907b-a6006ad3dba0" + +// uuidRegistry holds the two tables a malformed UUID takes records from: +// visits as a producer writes it (the id generated when omitted), and pings +// with the id first, for the positional formats. +func uuidRegistry(t *testing.T) *discovery.SchemaRegistry { + return testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ + { + Name: "visits", + Columns: []discovery.Column{ + {Name: "page", Type: "String", Position: 1}, + {Name: "id", Type: "UUID", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "generateUUIDv4()", Position: 2}, + }, + }, + { + Name: "pings", + Columns: []discovery.Column{ + {Name: "id", Type: "UUID", Position: 1}, + {Name: "page", Type: "String", Position: 2}, + {Name: "n", Type: "UInt8", Position: 3}, + }, + }, + }) +} + +// TestIngest_ShortAnswerDeclinesTheWholeBatch is the regression guard for a +// silent loss measured on 26.8.15.10: a UUID shorter than 36 characters makes +// ClickHouse's UUID reader consume a fixed 36-byte window past it, so the +// records the window reaches are never read and get no verdict. chtypes +// answered ONE verdict for these three-record bodies, and the batch reported +// `total: 1` with /b and /c gone. Now the body's own count holds: every record +// is answered declined and nothing is published. +func TestIngest_ShortAnswerDeclinesTheWholeBatch(t *testing.T) { + t.Parallel() + for _, tt := range []struct { + name, table, contentType, body string + }{ + {"a JSON array", "visits", "application/json", `[{"page":"/a","id":"x"},{"page":"/b"},{"page":"/c"}]`}, + {"an NDJSON body", "visits", "application/x-ndjson", "{\"page\":\"/a\",\"id\":\"x\"}\n{\"page\":\"/b\"}\n{\"page\":\"/c\"}\n"}, + {"a bare CSV body", "pings", "text/csv", "zzz,/a,1\n" + testUUID + ",/b,2\n" + testUUID + ",/c,3"}, + {"a header=absent CSV body", "pings", "text/csv; header=absent", "zzz,/a,1\n" + testUUID + ",/b,2\n" + testUUID + ",/c,3\n"}, + {"a header=present CSV body", "pings", "text/csv; header=present", "id,page,n\nzzz,/a,1\n" + testUUID + ",/b,2\n" + testUUID + ",/c,3\n"}, + {"a TSV body", "pings", "text/tab-separated-values", "zzz\t/a\t1\n" + testUUID + "\t/b\t2\n" + testUUID + "\t/c\t3\n"}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, uuidRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, tt.table, tt.contentType, tt.body))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 3, resp.Total, "total is the body's records, not chtypes' short count") + assert.Equal(t, 0, resp.Succeeded) + assert.Equal(t, 3, resp.Failed) + require.Len(t, resp.Results, 3) + for i, r := range resp.Results { + assert.Equal(t, i+1, r.Index) + assert.Contains(t, r.Error, "validation engine declined: the body holds") + assert.Contains(t, r.Error, "chtypes answered", "the message names both counts") + assert.Contains(t, r.Error, "record 1 was refused (code 376", "and where the records went missing") + assert.Zero(t, r.ExceptionCode, "a decline is not ClickHouse's verdict on the record") + } + assert.Empty(t, pub.Messages, "nothing from a short-answered body is published") + }) + } +} + +// TestIngest_ShortAnswerDeclinesTheSingleObject: a single object whose bad +// UUID swallows the objects concatenated after it is declined (422) rather +// than refused for its own value, because the body's count is not met. +// Alone, the same object is ClickHouse's own refusal. +func TestIngest_ShortAnswerDeclinesTheSingleObject(t *testing.T) { + t.Parallel() + h := newTestIngestHandler(t, uuidRegistry(t), &testutil.MockPublisher{}) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "visits", "application/json", `{"page":"/a","id":"x"}`))) + require.Equal(t, http.StatusBadRequest, w.Code, "body=%s", w.Body.String()) + + w = httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "visits", "application/json", `{"page":"/a","id":"x"} {"page":"/b"} {"page":"/c"}`))) + require.Equal(t, http.StatusUnprocessableEntity, w.Code, "body=%s", w.Body.String()) + assert.Contains(t, jsonErrorMessage(t, w), "the body holds at least 3 records but chtypes answered 1") +} + +// TestIngest_CountedBodiesKeepPerRecordVerdicts: the count check must never +// trip on a body chtypes read whole. A bad value that is not a short UUID is +// still one record's refusal with its siblings published, and quoted newlines, +// blank NDJSON lines, a pretty-printed array and every CSV header form keep +// their per-record answers. +func TestIngest_CountedBodiesKeepPerRecordVerdicts(t *testing.T) { + t.Parallel() + u := testUUID + for _, tt := range []struct { + name, table, contentType, body string + total, ok int + }{ + {"a bad UInt8 in a JSON array", "pings", "application/json", `[{"page":"/a","n":"bad"},{"page":"/b"},{"page":"/c"}]`, 3, 2}, + {"a full-length bad UUID in a JSON array", "pings", "application/json", `[{"page":"/a","id":"` + u[:35] + `z"},{"page":"/b"},{"page":"/c"}]`, 3, 2}, + {"a bad UUID in the last record", "pings", "application/x-ndjson", "{\"page\":\"/b\"}\n{\"page\":\"/a\",\"id\":\"x\"}\n", 2, 1}, + {"blank and whitespace NDJSON lines", "pings", "application/x-ndjson", "{\"page\":\"/a\"}\n\n \n{\"page\":\"/b\"}\n\n", 2, 2}, + {"a pretty-printed NDJSON body", "pings", "application/x-ndjson", "{\n \"page\": \"/a\"\n}\n{\n \"page\": \"/b\"\n}\n", 2, 2}, + {"a pretty-printed JSON array", "pings", "application/json", "[\n {\"page\": \"/a\"},\n {\"page\": \"/b\"}\n]\n", 2, 2}, + {"a raw newline inside a JSON string", "pings", "application/x-ndjson", "{\"page\":\"/a\nb\"}\n{\"page\":\"/c\"}\n", 2, 2}, + {"a quoted newline in bare CSV", "pings", "text/csv", u + ",\"/a\nx\",1\n" + u + ",/b,2\n", 2, 2}, + {"a quoted newline after a header", "pings", "text/csv; header=present", "page,id\n\"/a\nx\"," + u + "\n/b," + u + "\n", 2, 2}, + {"a detected names header", "pings", "text/csv", "id,page,n\n" + u + ",/a,1\n" + u + ",/b,2\n", 2, 2}, + {"a detected names and types header", "pings", "text/csv", "id,page,n\nUUID,String,UInt8\n" + u + ",/a,1\n", 1, 1}, + {"a detected subset header", "pings", "text/csv", "page,id\n/a," + u + "\n", 1, 1}, + {"a detected TSV header", "pings", "text/tab-separated-values", "id\tpage\tn\n" + u + "\t/a\t1\n", 1, 1}, + {"CRLF line endings", "pings", "text/csv; header=absent", u + ",/a,1\r\n" + u + ",/b,2\r\n", 2, 2}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, uuidRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, tt.table, tt.contentType, tt.body))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, tt.total, resp.Total) + assert.Equal(t, tt.ok, resp.Succeeded) + assert.Len(t, pub.Messages, tt.ok) + for _, r := range resp.Results { + assert.NotContains(t, r.Error, "declined", "a body chtypes read whole is never declined") + } + }) + } +} + +// TestIngest_EmptyArrayElementIsInvalidJSON: a leading, doubled or trailing +// comma frames as a blank line, which is no record to ClickHouse, so the +// element count could not hold. It is refused as the invalid JSON it is. +func TestIngest_EmptyArrayElementIsInvalidJSON(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, testRegistry(t), pub) + for _, body := range []string{`[{"page":"/a"},]`, `[{"page":"/a"},,{"page":"/b"}]`, `[,{"page":"/a"}]`} { + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "application/json", body))) + require.Equal(t, http.StatusBadRequest, w.Code, "body=%s", w.Body.String()) + assert.Contains(t, jsonErrorMessage(t, w), "invalid json: empty element in the json array") + } + assert.Empty(t, pub.Messages) +} diff --git a/internal/api/ingest_framing.go b/internal/api/ingest_framing.go index eac8ee17..ca0c2340 100644 --- a/internal/api/ingest_framing.go +++ b/internal/api/ingest_framing.go @@ -36,24 +36,29 @@ import ( // past it. JSONEachRow needs no brackets, so removing them costs nothing and // makes every position salvageable, first and last included; // - every other newline outside a string, before or after the brackets too, -// becomes a space. Not cosmetic: when chtypes declines a whole batch -// without per-record detail, the type layer counts the body's lines to -// answer each record, so a pretty-printed array would come back with one -// phantom declined record per line of layout. This leaves exactly -// elements-1 newlines, so that count stays right. +// becomes a space, leaving exactly elements-1 newlines. Not cosmetic: +// ClickHouse's reader resumes after a bad record at the next newline, so +// a layout newline inside an element would resume it mid-element. // // A raw newline inside a string is illegal JSON, so leaving those alone costs // nothing and keeps the caller's bytes the caller's. // +// The element count is exact, and it is what the type layer holds chtypes' +// answer to: any other number of verdicts declines the whole body +// (typelayer.IngestOptions.Records). +// // An error is a whole-request 400, and nothing is published from a body we // cannot frame: errUnterminatedArray when the brackets do not balance — a -// truncated upload, or a structural syntax error — and errAfterArray when +// truncated upload, or a structural syntax error — errEmptyElement for a +// leading, doubled or trailing comma, which frames as a blank line that is no +// record to ClickHouse and so would break the count, and errAfterArray when // anything but whitespace follows the array's closing ']'. That tail is not a // record of the array, and framing it as more records would publish what the // caller never put in the batch. func reframeArray(b []byte) (elements int, err error) { depth, commas := 0, 0 sawValue, closed := false, false + element := false // a value since the opening bracket or the last depth-1 comma inStr, esc := false, false for i := range b { c := b[i] @@ -75,9 +80,11 @@ func reframeArray(b []byte) (elements int, err error) { case c == '"': inStr = !inStr sawValue = sawValue || depth >= 1 + element = element || depth >= 1 case inStr: case c == '[' || c == '{': sawValue = sawValue || depth >= 1 + element = element || depth >= 1 depth++ if c == '[' && depth == 1 { b[i] = ' ' @@ -88,17 +95,25 @@ func reframeArray(b []byte) (elements int, err error) { if c != ']' { return 0, errUnterminatedArray // the array's '[' closed by a '}' } + if sawValue && !element { + return 0, errEmptyElement // `[{…},]` + } b[i] = ' ' closed = true } case c == ',' && depth == 1: + if !element { + return 0, errEmptyElement // `[,{…}]`, `[{…},,{…}]` + } b[i] = '\n' commas++ + element = false case c == '\n' || c == '\r': b[i] = ' ' case c == ' ' || c == '\t': default: sawValue = sawValue || depth >= 1 + element = element || depth >= 1 } } if depth != 0 || inStr { @@ -110,10 +125,11 @@ func reframeArray(b []byte) (elements int, err error) { return commas + 1, nil } -// The two ways reframeArray refuses a body, each the tail of the caller's +// The three ways reframeArray refuses a body, each the tail of the caller's // "invalid json: …" 400. var ( errUnterminatedArray = errors.New("unterminated json array") + errEmptyElement = errors.New("empty element in the json array (a leading, doubled or trailing comma)") errAfterArray = errors.New("content after the closing ']' of the json array") ) diff --git a/internal/api/ingest_framing_test.go b/internal/api/ingest_framing_test.go index f251c95b..226f3758 100644 --- a/internal/api/ingest_framing_test.go +++ b/internal/api/ingest_framing_test.go @@ -98,6 +98,11 @@ func TestReframeArray(t *testing.T) { {name: "a cut-off element does not balance", body: `[{"a":1},{"b`, err: errUnterminatedArray}, {name: "a structural syntax error does not balance", body: `[{"a":1}, {bad]`, err: errUnterminatedArray}, {name: "an array closed by a brace does not balance", body: `[{"a":1}}`, err: errUnterminatedArray}, + {name: "a trailing comma is an empty element", body: `[{"a":1},]`, err: errEmptyElement}, + {name: "a trailing comma before layout is an empty element", body: "[{\"a\":1},\n ]", err: errEmptyElement}, + {name: "a doubled comma is an empty element", body: `[{"a":1},,{"a":2}]`, err: errEmptyElement}, + {name: "a leading comma is an empty element", body: `[,{"a":1}]`, err: errEmptyElement}, + {name: "a lone comma is an empty element", body: `[,]`, err: errEmptyElement}, {name: "an object after the array is not a record of it", body: `[{"page":"a"},{"page":"b"}] {"page":"c","x":1}`, err: errAfterArray}, {name: "a second array after the first", body: `[{"a":1}][{"a":2}]`, err: errAfterArray}, {name: "a stray closing bracket after the array", body: `[{"a":1}]]`, err: errAfterArray}, diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go index 047eb96e..a2fcffae 100644 --- a/internal/typelayer/ingest.go +++ b/internal/typelayer/ingest.go @@ -28,6 +28,11 @@ type IngestOptions struct { // measured on the 26.6 and 26.8 servers and artifacts). Ignored for other // formats. StrictPositional bool + // Records is the caller's own exact count of the records in body, when its + // framing gives one (a JSON array's elements), and 0 otherwise. chtypes + // answering any other number declines the whole body; without it the type + // layer counts a floor itself (see recordFloor). + Records int } // parseSettings is what a body is parsed under: the insert pins, plus header @@ -89,15 +94,29 @@ type RowVerdict struct { // Batch holds one verdict per input record, in input order. type Batch struct { // Rows holds one verdict per record chtypes read, in input order. When - // Answered is false chtypes gave no per-record detail (the whole batch was - // declined) and Rows is padded to the body's line count so a caller still - // has something index-shaped to report. + // Answered is false the whole batch was declined — chtypes gave no + // per-record detail, or answered fewer records than the body holds + // (Miscount) — and Rows holds one declined verdict per record counted in + // the body, so a caller still has something index-shaped to report. Rows []RowVerdict Answered bool // Refused is ClickHouse's own refusal of a WithNames body as a whole — a // header naming a column the schema does not have, or naming one twice — // before any record was read. Rows is then empty; nil otherwise. Refused *Refusal + // Miscount is set when chtypes answered a different number of records than + // the body holds: fewer than the type layer's own floor, or other than the + // caller's exact count (IngestOptions.Records). The batch is then declined + // whole — Answered false, one declined verdict per counted record — so a + // record chtypes never read cannot go unreported. nil otherwise. + Miscount *Miscount +} + +// Miscount is how far chtypes' answer fell from the body's own count. +type Miscount struct { + Counted int // records in the body: the caller's exact count, or a floor + Verdicts int // verdicts chtypes returned + Exact bool // Counted is IngestOptions.Records rather than a floor } // Refusal is ClickHouse's verdict on a body rather than on any one record. @@ -137,6 +156,10 @@ type Refusal struct { // accepted record. The compiled filter is cached per (generation, expression, // values) on the handle it runs against. // +// The verdicts account for every record or for none: when chtypes answers +// fewer records than the body holds (recordFloor), the whole body is declined +// (Batch.Miscount) rather than returned short. +// // Document flags stay lean (verdicts and exported bytes only). The per-value // provenance DocValues would give costs 2.65× on this path and nothing here // reads it. @@ -187,18 +210,27 @@ func (t *Table) IngestWith(format Format, opts IngestOptions, body []byte, check (format == FormatCSVWithNames || format == FormatTSVWithNames) { return Batch{Answered: true, Refused: &Refusal{Code: res.ErrCode, Message: res.ErrMsg}}, nil } - return declineAll(countRecords(body, len(res.Rows)), firstNonEmpty(res.ExportDeclined, res.ErrMsg, res.Outcome.String())), nil + n := declineCount(opts.Records, len(res.Rows), t.floor(s, format, opts, body), body) + return declineAll(n, firstNonEmpty(res.ExportDeclined, res.ErrMsg, res.Outcome.String())), nil } // Accepted but withheld (the full-arity guard, a serialization failure): // nothing can be forwarded, whatever the per-row detail says. if res.ExportDeclined != "" { - return declineAll(countRecords(body, len(res.Rows)), res.ExportDeclined), nil + n := declineCount(opts.Records, len(res.Rows), t.floor(s, format, opts, body), body) + return declineAll(n, res.ExportDeclined), nil } - // chtypes answered per record, so its count is the record count: the - // verdicts are index-aligned with the records it read, and padding to the - // body's newline count would invent declined records out of blank lines - // and pretty-printed framing. + // chtypes answered per record, and its verdicts are index-aligned with the + // records it read — which is every record only if it read as many as the + // body holds. Fewer means its reader took some records with another (see + // recordFloor), and reporting the short batch would drop them without a + // word: decline the body whole instead. More is checked only against an + // exact count, because a floor is no ceiling. + if m := miscount(opts.Records, len(res.Rows), t.floor(s, format, opts, body)); m != nil { + b := declineAll(m.Counted, m.message(res.Rows)) + b.Miscount = m + return b, nil + } out := Batch{Rows: make([]RowVerdict, len(res.Rows)), Answered: true} for i := range out.Rows { v := rowVerdict(res.Rows[i], span(res, i), filter != nil) @@ -309,6 +341,64 @@ func span(res chtypes.BatchResult, i int) []byte { return bytes.TrimSuffix(res.Payload[s.Off:s.Off+s.Len], []byte("\n")) } +// floor is recordFloor for body against this handle's columns: header +// auto-detection compares a first line with the wire columns and a second with +// their compiled types. Skipped when the caller has an exact count. +func (t *Table) floor(s *schemaSlot, format Format, opts IngestOptions, body []byte) int { + if opts.Records > 0 { + return 0 + } + var types map[string]string + if (format == FormatCSV || format == FormatTSV) && !opts.StrictPositional { + types = make(map[string]string, len(s.schema.Columns)) + for _, c := range s.schema.Columns { + types[c.Name] = c.Type + } + } + return recordFloor(format, opts, body, t.WireColumns, types) +} + +// miscount compares chtypes' verdict count with the body's: exact (the +// caller's count, any difference) or a floor (fewer only). nil when they agree. +func miscount(exact, verdicts, floor int) *Miscount { + switch { + case exact > 0 && verdicts != exact: + return &Miscount{Counted: exact, Verdicts: verdicts, Exact: true} + case exact == 0 && verdicts < floor: + return &Miscount{Counted: floor, Verdicts: verdicts} + } + return nil +} + +// message is every declined record's error: the two counts, and the first +// record chtypes refused, which is where the records went missing — the +// verdicts before it are aligned with the body, so its index is the caller's. +func (m *Miscount) message(rows []chtypes.RowResult) string { + holds := "at least " + if m.Exact { + holds = "" + } + msg := fmt.Sprintf("the body holds %s%d records but chtypes answered %d, so the batch is declined whole rather than reported short", + holds, m.Counted, m.Verdicts) + for i, r := range rows { + if r.Outcome == chtypes.Rejected || r.Outcome == chtypes.Skipped { + return msg + fmt.Sprintf("; record %d was refused (code %d: %s) and its reader may have taken the records after it", i+1, r.ErrCode, r.ErrMsg) + } + } + return msg +} + +// declineCount is how many records a whole-batch decline answers: the caller's +// exact count when it has one, else the larger of chtypes' own count and the +// floor — chtypes may have stopped short of the body's end — and the body's +// line count when neither saw a record. +func declineCount(exact, verdicts, floor int, body []byte) int { + if exact > 0 { + return exact + } + return countRecords(body, max(verdicts, floor)) +} + func declineAll(n int, msg string) Batch { b := Batch{Rows: make([]RowVerdict, n)} for i := range b.Rows { @@ -317,14 +407,12 @@ func declineAll(n int, msg string) Batch { return b } -// countRecords recovers the input record count when chtypes returned no -// per-row detail, so the caller still gets an index-aligned answer. JSONEachRow -// records are newline-separated and a raw newline inside a JSON string is -// illegal, so counting lines is exact for compact NDJSON and a re-framed array. -// A blank line, a pretty-printed object's inner lines, a CSV field holding a -// raw newline and a WithNames header each add a line that is no record, so the -// fallback can OVER-count, which produces extra declined verdicts — never an -// extra acceptance. +// countRecords is the record count of a declined batch: known when anything +// counted a record, else the body's line count, so the caller still gets an +// index-shaped answer. A blank line, a pretty-printed object's inner lines, a +// CSV field holding a raw newline and a WithNames header each add a line that +// is no record, so the fallback can OVER-count, which produces extra declined +// verdicts — never an extra acceptance. func countRecords(body []byte, known int) int { if known > 0 { return known diff --git a/internal/typelayer/records.go b/internal/typelayer/records.go new file mode 100644 index 00000000..21f47640 --- /dev/null +++ b/internal/typelayer/records.go @@ -0,0 +1,370 @@ +package typelayer + +import ( + "bytes" + "strconv" +) + +// Counting the records a body holds, independently of chtypes. +// +// chtypes answers one verdict per record its reader took, and its reader can +// take more than one record at a time. Measured on 26.8.15.10: a UUID value +// shorter than 36 characters makes the UUID reader consume a fixed 36-byte +// window past it, and the error recovery that follows resumes at the end of the +// line the window ended on — so the records the window reached are gone from +// the answer, with no verdict of their own. `{"id":"x"}` followed by two short +// records answers ONE verdict; the same in CSV loses the line after. Nothing +// in the per-record detail says so: the batch outcome is still Accepted. +// +// So the type layer counts the records itself and refuses to report a short +// batch: fewer verdicts than recordFloor means records went unread, and the +// whole body is declined (see IngestWith). The count is a FLOOR, never an +// estimate, so a body chtypes read whole can never trip it: +// +// - JSONEachRow: top-level objects, scanned string- and escape-aware — a +// raw newline inside a string is part of the string to ClickHouse too +// (measured), and blank lines, pretty-printing, commas between objects and +// an outer array are not records. Exact for a well-formed body; a malformed +// one can only under-count (a truncated object hides the ones after it), so +// a scalar or garbage line costs the check its precision, never a false +// decline. +// - CSV and TSV: records, which to ClickHouse are LINES — a blank line is a +// record (an empty value on a one-column table, a refused record on a +// wider one; measured), as is a non-empty tail with no newline. A CSV +// newline inside a double-quoted field, a quote opening after spaces or +// tabs included, is not a terminator; a quote anywhere else is a literal. +// A TSV newline escaped by a backslash is not one either. The header line +// of a WithNames body is not a record; under header auto-detection see +// detectedHeaderRows. +func recordFloor(format Format, opts IngestOptions, body []byte, wire []string, types map[string]string) int { + switch format { + case FormatJSONEachRow: + return jsonObjects(body) + case FormatCSV, FormatCSVWithNames, FormatTSV, FormatTSVWithNames: + default: + return 0 + } + csv := format == FormatCSV || format == FormatCSVWithNames + var n int + var first, second []byte + if csv { + n, first, second = csvRecords(body) + } else { + n, first, second = tsvRecords(body) + } + switch { + case n == 0: + return 0 + case format == FormatCSVWithNames || format == FormatTSVWithNames: + // Exactly one header line; a second line spelling the types is a + // record (measured). + return n - 1 + case opts.StrictPositional: + return n + default: + return n - detectedHeaderRows(csv, first, second, n, wire, types) + } +} + +// jsonObjects counts the top-level objects of a JSONEachRow body: those that +// open at depth 0, or at depth 1 inside an outer array. +func jsonObjects(b []byte) int { + n, depth := 0, 0 + inArray, inStr, esc := false, false, false + for _, c := range b { + switch { + case esc: + esc = false + case inStr && c == '\\': + esc = true + case c == '"': + inStr = !inStr + case inStr: + case c == '{' || c == '[': + if depth == 0 && c == '[' { + inArray = true + } else if c == '{' && (depth == 0 || depth == 1 && inArray) { + n++ + } + depth++ + case c == '}' || c == ']': + if depth > 0 { + depth-- + } + if depth == 0 { + inArray = false + } + } + } + return n +} + +// csvRecords counts ClickHouse's CSV records in b and returns the first two, +// without their terminators. A double quote opens a quoted field only where a +// field starts (leading spaces and tabs aside); inside one, `""` is a quote and +// a newline is data. ClickHouse reads the same body the same way (measured, +// including a quote after leading whitespace, a quote mid-field and a CRLF +// inside quotes). +func csvRecords(b []byte) (n int, first, second []byte) { + start := 0 + inQuote, fieldStart := false, true + for i := 0; i < len(b); i++ { + c := b[i] + if inQuote { + if c == '"' { + if i+1 < len(b) && b[i+1] == '"' { + i++ + continue + } + inQuote = false + } + continue + } + switch c { + case '"': + inQuote = fieldStart + fieldStart = false + case ',': + fieldStart = true + case ' ', '\t': + case '\n': + n, first, second = takeRecord(n, first, second, b[start:i]) + start, fieldStart = i+1, true + default: + fieldStart = false + } + } + if start < len(b) { + n, first, second = takeRecord(n, first, second, b[start:]) + } + return n, first, second +} + +// tsvRecords is csvRecords for TSV: no quoting, and a newline after an odd run +// of backslashes is an escaped newline inside a field (measured: `a\` + LF is +// one value, `a\\` + LF ends the record). +func tsvRecords(b []byte) (n int, first, second []byte) { + start, run := 0, 0 + for i, c := range b { + if c == '\n' && run%2 == 0 { + n, first, second = takeRecord(n, first, second, b[start:i]) + start = i + 1 + } + if c == '\\' { + run++ + } else { + run = 0 + } + } + if start < len(b) { + n, first, second = takeRecord(n, first, second, b[start:]) + } + return n, first, second +} + +func takeRecord(n int, first, second, rec []byte) (int, []byte, []byte) { + switch n { + case 0: + first = rec + case 1: + second = rec + } + return n + 1, first, second +} + +// detectedHeaderRows is how many leading records ClickHouse's header +// auto-detection (input_format_{csv,tsv}_detect_header, on for bare CSV and +// TSV) consumes, as measured on 26.8.15.10: +// +// - the first record is a names row when its values, as a set, are +// contained in the table's columns or contain them — case-sensitive, CSV +// values unquoted and trimmed of spaces and tabs, TSV values unescaped and +// not trimmed; +// - after a names row, the second record is consumed too when it spells +// column types. Accepted only when each value is exactly the type of the +// column the names row put there; any other set of valid type names +// rejects the whole batch, which is declined without reaching this count. +// +// Where a value cannot be read the way ClickHouse reads it (an escape this does +// not decode, an unterminated quote), it is taken to match, so the answer +// errs towards consuming a row: that lowers the floor, which can only miss a +// lost record, never decline a body chtypes read whole. +func detectedHeaderRows(csv bool, first, second []byte, n int, wire []string, types map[string]string) int { + names, known := splitRecord(csv, first) + if !namesRow(names, known, wire) { + return 0 + } + if n < 2 { + return 1 + } + values, vknown := splitRecord(csv, second) + if typesRow(names, values, vknown, types) { + return 2 + } + return 1 +} + +// namesRow reports whether values could be a names row for columns wire. A +// value whose spelling is not known (known[i] false) may be any column. +func namesRow(values []string, known []bool, wire []string) bool { + cols := make(map[string]struct{}, len(wire)) + for _, c := range wire { + cols[c] = struct{}{} + } + seen := make(map[string]struct{}, len(values)) + inCols, wild := true, 0 + for i, v := range values { + if !known[i] { + wild++ + continue + } + seen[v] = struct{}{} + if _, ok := cols[v]; !ok { + inCols = false + } + } + if inCols { + return true + } + missing := 0 + for c := range cols { + if _, ok := seen[c]; !ok { + missing++ + } + } + return missing <= wild +} + +// typesRow reports whether values spells, position by position, the types of +// the columns names put there. +func typesRow(names, values []string, known []bool, types map[string]string) bool { + if len(values) != len(names) { + return false + } + for i, v := range values { + if !known[i] { + continue + } + if t, ok := types[names[i]]; !ok || t != v { + return false + } + } + return true +} + +// splitRecord splits one record into the values header detection compares, +// with known[i] false where value i's spelling could not be decoded exactly. +func splitRecord(csv bool, rec []byte) (values []string, known []bool) { + if csv { + return csvValues(rec) + } + return tsvValues(rec) +} + +func csvValues(rec []byte) (values []string, known []bool) { + rec = bytes.TrimSuffix(rec, []byte("\r")) + for { + rec = bytes.TrimLeft(rec, " \t") + var v []byte + ok := true + if len(rec) > 0 && rec[0] == '"' { + j := 1 + for ; j < len(rec); j++ { + if rec[j] != '"' { + v = append(v, rec[j]) + continue + } + if j+1 < len(rec) && rec[j+1] == '"' { + v = append(v, '"') + j++ + continue + } + break + } + if j >= len(rec) { + return append(values, string(v)), append(known, false) // unterminated + } + rec = bytes.TrimLeft(rec[j+1:], " \t") + if len(rec) > 0 && rec[0] != ',' { + // Text after the closing quote: ClickHouse refuses the field. + ok = false + k := bytes.IndexByte(rec, ',') + if k < 0 { + k = len(rec) + } + rec = rec[k:] + } + } else { + k := bytes.IndexByte(rec, ',') + if k < 0 { + k = len(rec) + } + v = bytes.TrimRight(rec[:k], " \t") + rec = rec[k:] + } + values, known = append(values, string(v)), append(known, ok) + if len(rec) == 0 { + return values, known + } + rec = rec[1:] // the comma + } +} + +func tsvValues(rec []byte) (values []string, known []bool) { + for f := range bytes.SplitSeq(rec, []byte("\t")) { + v, ok := tsvUnescape(f) + values, known = append(values, v), append(known, ok) + } + return values, known +} + +// tsvUnescape decodes TabSeparated's escapes. ok is false for one it does not +// decode, or a trailing CR (whose reading depends on a CRLF setting). +func tsvUnescape(f []byte) (string, bool) { + if bytes.HasSuffix(f, []byte("\r")) { + return "", false + } + if bytes.IndexByte(f, '\\') < 0 { + return string(f), true + } + out := make([]byte, 0, len(f)) + for i := 0; i < len(f); i++ { + if f[i] != '\\' { + out = append(out, f[i]) + continue + } + if i+1 >= len(f) { + return "", false + } + i++ + switch f[i] { + case '\\', '\'', '"': + out = append(out, f[i]) + case 't': + out = append(out, '\t') + case 'n', '\n': + out = append(out, '\n') + case 'r': + out = append(out, '\r') + case 'b': + out = append(out, '\b') + case 'f': + out = append(out, '\f') + case '0': + out = append(out, 0) + case 'x': + if i+2 >= len(f) { + return "", false + } + h, err := strconv.ParseUint(string(f[i+1:i+3]), 16, 8) + if err != nil { + return "", false + } + out = append(out, byte(h)) + i += 2 + default: + return "", false + } + } + return string(out), true +} diff --git a/internal/typelayer/records_test.go b/internal/typelayer/records_test.go new file mode 100644 index 00000000..e0b33a99 --- /dev/null +++ b/internal/typelayer/records_test.go @@ -0,0 +1,265 @@ +package typelayer + +import ( + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/wave-rf/chtypes/go/chtypes" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +const recUUID = "61f0c404-5cb3-11e7-907b-a6006ad3dba0" + +// uuidTables are a UUID-first table for the positional formats and a +// name-addressed one whose omitted id takes a constant DEFAULT. +func uuidTables() []*discovery.TableSchema { + return []*discovery.TableSchema{ + {Name: "pings", Columns: []discovery.Column{ + {Name: "id", Type: "UUID", Position: 1}, + {Name: "page", Type: "String", Position: 2}, + {Name: "n", Type: "UInt8", Position: 3}, + }}, + {Name: "visits", Columns: []discovery.Column{ + {Name: "page", Type: "String", Position: 1}, + {Name: "id", Type: "UUID", HasDefault: true, DefaultKind: "DEFAULT", DefaultExpression: "toUUID('00000000-0000-0000-0000-000000000000')", Position: 2}, + }}, + } +} + +// TestIngest_ShortAnswerIsDeclinedWhole pins the measured loss at its source: +// a UUID shorter than 36 characters makes ClickHouse's reader consume a fixed +// 36-byte window past it, and the records that window reaches get no verdict +// while the batch outcome stays Accepted. IngestWith must answer every record +// the body holds, declined, rather than return the short batch. +func TestIngest_ShortAnswerIsDeclinedWhole(t *testing.T) { + eng := testEngine(t, uuidTables()...) + table := func(name string) *Table { + tbl, err := eng.Table(tenant.Default, name) + require.NoError(t, err) + t.Cleanup(tbl.Release) + return tbl + } + visits, pings := table("visits"), table("pings") + + ndjson6 := `{"page":"/a","id":"x"}` + "\n" + strings.Repeat(`{"page":"/b"}`+"\n", 5) + for _, tt := range []struct { + name string + tbl *Table + format Format + opts IngestOptions + body string + want int + }{ + // chtypes answered 3 of these six: the bad record and the two the + // window did not reach. + {"six NDJSON records", visits, FormatJSONEachRow, IngestOptions{}, ndjson6, 6}, + {"three NDJSON records", visits, FormatJSONEachRow, IngestOptions{}, "{\"page\":\"/a\",\"id\":\"x\"}\n{\"page\":\"/b\"}\n{\"page\":\"/c\"}", 3}, + {"a caller's exact count", visits, FormatJSONEachRow, IngestOptions{Records: 3}, " {\"page\":\"/a\",\"id\":\"x\"}\n{\"page\":\"/b\"}\n{\"page\":\"/c\"} ", 3}, + {"bare CSV", pings, FormatCSV, IngestOptions{}, "zzz,/a,1\n" + recUUID + ",/b,2\n" + recUUID + ",/c,3\n", 3}, + // A header line under header=absent is a record too, and its column + // name is a short UUID. + {"a header line read as a record", pings, FormatCSV, IngestOptions{StrictPositional: true}, "id,page,n\n" + recUUID + ",/b,2\n", 2}, + {"a blank CSV line", pings, FormatCSV, IngestOptions{StrictPositional: true}, recUUID + ",/a,1\n\n" + recUUID + ",/b,2\n", 3}, + {"a CSVWithNames types line", pings, FormatCSVWithNames, IngestOptions{}, "id,page,n\nUUID,String,UInt8\n" + recUUID + ",/b,2\n", 2}, + {"TSV", pings, FormatTSV, IngestOptions{}, "zzz\t/a\t1\n" + recUUID + "\t/b\t2\n" + recUUID + "\t/c\t3\n", 3}, + } { + t.Run(tt.name, func(t *testing.T) { + batch, err := tt.tbl.IngestWith(tt.format, tt.opts, []byte(tt.body)) + require.NoError(t, err) + require.NotNil(t, batch.Miscount, "chtypes answered short; the batch must say so") + assert.Equal(t, tt.want, batch.Miscount.Counted) + assert.Less(t, batch.Miscount.Verdicts, tt.want) + assert.Equal(t, tt.opts.Records > 0, batch.Miscount.Exact) + assert.False(t, batch.Answered) + require.Len(t, batch.Rows, tt.want, "one verdict per record the body holds") + for _, r := range batch.Rows { + assert.True(t, r.Declined, "declined, never accepted or refused: %+v", r) + assert.Nil(t, r.Line) + assert.Contains(t, r.Message, "chtypes answered") + } + }) + } +} + +// TestIngest_WholeAnswersAreNotMiscounted: bodies chtypes reads whole keep +// their per-record answers — the count is a floor and must never trip on +// them. +func TestIngest_WholeAnswersAreNotMiscounted(t *testing.T) { + eng := testEngine(t, uuidTables()...) + table := func(name string) *Table { + tbl, err := eng.Table(tenant.Default, name) + require.NoError(t, err) + t.Cleanup(tbl.Release) + return tbl + } + visits, pings := table("visits"), table("pings") + r := recUUID + ",/r,1\n" + for _, tt := range []struct { + name string + tbl *Table + format Format + opts IngestOptions + body string + rows int + }{ + {"a bad UUID in the last record", visits, FormatJSONEachRow, IngestOptions{}, "{\"page\":\"/b\"}\n{\"page\":\"/a\",\"id\":\"x\"}\n", 2}, + {"a 36-character bad UUID", visits, FormatJSONEachRow, IngestOptions{}, "{\"page\":\"/a\",\"id\":\"" + recUUID[:35] + "z\"}\n{\"page\":\"/b\"}\n", 2}, + {"concatenated, comma-separated and bracketed objects", visits, FormatJSONEachRow, IngestOptions{}, `{"page":"/a"}{"page":"/b"},{"page":"/c"}`, 3}, + {"an outer array", visits, FormatJSONEachRow, IngestOptions{}, `[{"page":"/a"},{"page":"/b"}]`, 2}, + {"blank and whitespace lines", visits, FormatJSONEachRow, IngestOptions{}, "{\"page\":\"/a\"}\n \n\t\n{\"page\":\"/b\"}\n", 2}, + {"a raw newline inside a string", visits, FormatJSONEachRow, IngestOptions{}, "{\"page\":\"a\n{\"}\n{\"page\":\"/c\"}\n", 2}, + {"a scalar line", visits, FormatJSONEachRow, IngestOptions{}, "1\n{\"page\":\"/b\"}\n", 2}, + {"a quoted CSV newline", pings, FormatCSV, IngestOptions{StrictPositional: true}, recUUID + ",\"/a\nx\",1\n" + r, 2}, + {"a quoted CSV newline after whitespace", pings, FormatCSV, IngestOptions{StrictPositional: true}, recUUID + ", \"/a\nx\",1\n" + r, 2}, + {"a mid-field quote", pings, FormatCSV, IngestOptions{StrictPositional: true}, recUUID + ",a\"b,1\n" + r, 2}, + {"CRLF", pings, FormatCSV, IngestOptions{StrictPositional: true}, recUUID + ",/a,1\r\n" + recUUID + ",/b,2\r\n", 2}, + {"an escaped TSV newline", pings, FormatTSV, IngestOptions{StrictPositional: true}, recUUID + "\t/a\\\nx\t1\n" + recUUID + "\t/b\t2\n", 2}, + {"a detected names header", pings, FormatCSV, IngestOptions{}, "id,page,n\n" + r, 1}, + {"a detected quoted, padded header", pings, FormatCSV, IngestOptions{}, " \"id\", page , n\n" + r, 1}, + {"a detected subset header", pings, FormatCSV, IngestOptions{}, "page,id\n/r," + recUUID + "\n", 1}, + {"a detected names and types header", pings, FormatCSV, IngestOptions{}, "page,id,n\nString,UUID,UInt8\n/r," + recUUID + ",1\n", 1}, + {"a detected TSV header with an escape", pings, FormatTSV, IngestOptions{}, "i\\x64\tpage\tn\n" + recUUID + "\t/r\t1\n", 1}, + {"a CSVWithNames header", pings, FormatCSVWithNames, IngestOptions{}, "page,id\n/a," + recUUID + "\n/b," + recUUID + "\n", 2}, + {"a header with no records", pings, FormatCSV, IngestOptions{}, "id,page,n\n", 0}, + } { + t.Run(tt.name, func(t *testing.T) { + batch, err := tt.tbl.IngestWith(tt.format, tt.opts, []byte(tt.body)) + require.NoError(t, err) + assert.Nil(t, batch.Miscount) + assert.True(t, batch.Answered) + assert.Len(t, batch.Rows, tt.rows) + }) + } +} + +func TestJSONObjects(t *testing.T) { + t.Parallel() + for body, want := range map[string]int{ + ``: 0, + `{}`: 1, + "{\"a\":1}\n{\"a\":2}\n": 2, + `{"a":{"b":{}}}{"c":[{},{}]}`: 2, + `[{"a":1},{"a":2}]`: 2, + `{"a":"}{"}`: 1, + `{"a":"\"}{"}`: 1, + "{\n \"a\": 1\n}\n\n{\n \"a\": 2\n}": 2, + `{"a":1}}{"b":2}`: 2, // a stray closer does not hide what follows + "{\"a\":1\n{\"b\":2}\n": 1, // a truncated object hides what follows: a floor, not a count + "1\n\"x\"\nnull\n": 0, + } { + assert.Equal(t, want, jsonObjects([]byte(body)), "%q", body) + } +} + +func TestCSVRecords(t *testing.T) { + t.Parallel() + for _, tt := range []struct { + body string + n int + first, second string + }{ + {"", 0, "", ""}, + {"a", 1, "a", ""}, + {"a\n", 1, "a", ""}, + {"a\nb", 2, "a", "b"}, + {"\n", 1, "", ""}, + {"a\n\nb\n", 3, "a", ""}, + {"a\n ", 2, "a", " "}, + {"\"x\ny\",1\nb\n", 2, "\"x\ny\",1", "b"}, + {" \"x\ny\",1\nb\n", 2, " \"x\ny\",1", "b"}, + {"\t\"x\ny\"\nb\n", 2, "\t\"x\ny\"", "b"}, + {"\"a\"\"\nb\"\nc\n", 2, "\"a\"\"\nb\"", "c"}, + {"\"a\\\"\nb\"\nc\n", 3, "\"a\\\"", "b\""}, + {"a\"b\nc\",1\n", 2, "a\"b", "c\",1"}, + {"'a\nb',1\n", 2, "'a", "b',1"}, + {"a\r\nb\r\n", 2, "a\r", "b\r"}, + {"\"never closed\nstill quoted", 1, "\"never closed\nstill quoted", ""}, + } { + n, first, second := csvRecords([]byte(tt.body)) + assert.Equal(t, tt.n, n, "%q", tt.body) + assert.Equal(t, tt.first, string(first), "%q first", tt.body) + assert.Equal(t, tt.second, string(second), "%q second", tt.body) + } +} + +func TestTSVRecords(t *testing.T) { + t.Parallel() + for body, want := range map[string]int{ + "": 0, + "a\tb\n": 1, + "a\\\nb\nc\n": 2, + "a\\\\\nb\nc\n": 3, + "a\\\\\\\nb\nc\n": 2, + "a\n\nb": 3, + "\"a\nb\"\n": 2, + } { + n, _, _ := tsvRecords([]byte(body)) + assert.Equal(t, want, n, "%q", body) + } +} + +// TestDetectedHeaderRows pins the header auto-detection rule measured on +// 26.8.15.10, and the direction it errs in where a value cannot be decoded. +func TestDetectedHeaderRows(t *testing.T) { + t.Parallel() + wire := []string{"id", "page", "n"} + types := map[string]string{"id": "UUID", "page": "String", "n": "UInt8"} + for _, tt := range []struct { + name string + csv bool + first, second string + n, want int + }{ + {"names", true, "id,page,n", recUUID + ",/a,1", 2, 1}, + {"names, quoted and padded", true, ` "id" , page,n `, "x", 2, 1}, + {"a subset of the names", true, "page", "x", 2, 1}, + {"a superset of the names", true, "id,page,n,extra", "x", 2, 1}, + {"case differs", true, "ID,Page,N", "x", 2, 0}, + {"types alone", true, "UUID,String,UInt8", "x", 2, 0}, + {"data", true, "zzz,/a,1", "x", 2, 0}, + {"names then types", true, "id,page,n", "UUID,String,UInt8", 3, 2}, + {"names then types in the header's order", true, "page,id", "String,UUID", 3, 2}, + {"names then other identifiers", true, "id,page,n", "Foo,Bar,Baz", 3, 1}, + {"names then a type alias", true, "id,page,n", "UUID,TEXT,UInt8", 3, 1}, + {"names alone", true, "id,page,n", "", 1, 1}, + {"a CRLF names line", true, "id,page,n\r", "x", 2, 1}, + {"an unterminated quote may be a name", true, `id,page,"n`, "x", 2, 1}, + {"TSV names", false, "id\tpage\tn", "x", 2, 1}, + {"TSV names are not trimmed", false, " id \tpage", "x", 2, 0}, + {"TSV names unescape", false, "i\\x64\tpage\\\\\tn", "x", 2, 0}, + {"TSV hex escape", false, "i\\x64\tpage\tn", "x", 2, 1}, + {"an escape not decoded may be a name", false, "\\N\tpage\tn", "x", 2, 1}, + {"a TSV CR may be anything", false, "id\tpage\tn\r", "x", 2, 1}, + } { + got := detectedHeaderRows(tt.csv, []byte(tt.first), []byte(tt.second), tt.n, wire, types) + assert.Equal(t, tt.want, got, tt.name) + } +} + +func TestRecordFloor(t *testing.T) { + t.Parallel() + wire := []string{"id", "page", "n"} + types := map[string]string{"id": "UUID", "page": "String", "n": "UInt8"} + body := []byte("id,page,n\na\nb\n") + assert.Equal(t, 2, recordFloor(FormatCSV, IngestOptions{}, body, wire, types), "detected header") + assert.Equal(t, 3, recordFloor(FormatCSV, IngestOptions{StrictPositional: true}, body, wire, types), "header=absent") + assert.Equal(t, 2, recordFloor(FormatCSVWithNames, IngestOptions{}, body, wire, types), "header=present") + assert.Equal(t, 0, recordFloor(FormatCSVWithNames, IngestOptions{}, nil, wire, types)) + assert.Equal(t, 2, recordFloor(FormatJSONEachRow, IngestOptions{}, []byte("{}\n{}\n"), nil, nil)) + assert.Equal(t, 0, recordFloor(chtypes.JSONCompactEachRow, IngestOptions{}, []byte("[1]\n"), nil, nil), "a format Ingest does not parse") +} + +func TestMiscount(t *testing.T) { + t.Parallel() + assert.Nil(t, miscount(0, 3, 3)) + assert.Nil(t, miscount(0, 4, 3), "a floor is no ceiling") + assert.Equal(t, &Miscount{Counted: 3, Verdicts: 1}, miscount(0, 1, 3)) + assert.Nil(t, miscount(3, 3, 0)) + assert.Equal(t, &Miscount{Counted: 3, Verdicts: 4, Exact: true}, miscount(3, 4, 0), "an exact count holds both ways") + assert.Equal(t, &Miscount{Counted: 3, Verdicts: 1, Exact: true}, miscount(3, 1, 0)) +} From f0213ee0d513e9038fe57e2bee7698fe0a3b4fdc Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 13:52:40 -0400 Subject: [PATCH 69/70] docs: correct the type layer's unavailable causes, NaN rendering and literals - deployment.md: a tenant not bound yet is not like the other three causes. The type layer is bound in the refresh that marks the schema loaded, so its ingest answers `schema not loaded yet`, and the probes report it while no tenant has completed a discovery. api.md's two `ingest validation is unavailable` rows say the same. - deployment.md: another ClickHouse line, or a second server time zone on one line, refuses the tenant's ingest and withholds its row-filtered streams; it does not refuse the tenant. "Five" consequences were four. - NaN and infinity: /v1/query and pipes render null, the stream a string. The export ignores output_format_json_quote_denormals (measured on 26.8.15.10), and its string is what the worker's INSERT stores as the value where a null would store the column default, so the surfaces are scoped in the docs rather than pinned; a test pins the export spelling. - access-control.mdx: "1.0" is a legal Decimal, so the example of a literal another numeric column cannot read is "abc" (53 on a Decimal, 72 on a Float, measured on a 26.8.15.10 server). - api.md: 117 is also too many fields on a positional line. - A source build links against the build host's glibc; 2.34 is the prebuilt binaries' floor. ClickHouse is the only external service. Co-Authored-By: Claude Opus 5.5 --- README.md | 2 +- docs/src/content/docs/access-control.mdx | 4 ++-- docs/src/content/docs/api.md | 8 ++++---- docs/src/content/docs/deployment.md | 10 +++++----- docs/src/content/docs/development.md | 4 ++-- docs/src/content/docs/getting-started.md | 2 +- docs/src/content/docs/sdk/reference.md | 2 +- internal/typelayer/ingest_test.go | 24 ++++++++++++++++++++++++ 8 files changed, 40 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index 4f1d0d6d..7c8a1348 100644 --- a/README.md +++ b/README.md @@ -129,7 +129,7 @@ The published images bake the chtypes artifact for ClickHouse 26.8 only. Against go install github.com/Wave-RF/WaveHouse/cmd/wavehouse@latest ``` -`go install` compiles from source with cgo enabled (requires a C toolchain, and on Linux glibc 2.34 or later — Linux amd64/arm64 or macOS arm64) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse needs: the server refuses to boot without one. Fetch it once before the first run: +`go install` compiles from source with cgo enabled (requires a C toolchain — Linux amd64/arm64 or macOS arm64; on Linux it links against the build host's glibc, where the prebuilt binaries need 2.34 or later) but does not fetch the [chtypes artifact](https://wavehouse.dev/deployment#chtypes-artifacts) WaveHouse needs: the server refuses to boot without one. Fetch it once before the first run: ```bash go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index a7103df2..fb14bbe0 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -241,7 +241,7 @@ Any `filter` or `check` operator value may interpolate token claims with `{{ jwt - `{{ jwt.sub }}` → the token's `sub` claim. - `{{ jwt.app_metadata.tenant_id }}` → a nested claim. -Values are always bound as SQL **parameters**, never concatenated into the query, so templating is injection-safe. If a claim path in a `filter` template can't be resolved (a validly-signed token that simply doesn't carry the claim), that filter **fails closed**: the predicate becomes constant-false, so on the structured-query path (`POST /v1/query`) the role sees **no rows**, and the live stream withholds every event for that subscriber — one resolution drives both read surfaces (see [where each rule is enforced](#where-each-rule-is-enforced)). This holds for every operator — `_eq`, `_neq`, `_gt`, `_lt`, and `_in` alike. The alternative, binding the empty string the template would render to, would leave a live predicate against `''`: `_eq` would match every empty-valued row, and `_neq`/`_gt` on a string column would match essentially *all* rows, erasing the restriction. A literal value with no template in it — including an explicit `""` — binds exactly as written, and ClickHouse reads it under the column's type: a `String` column compares that exact text, while a numeric column reads the constant as a number. A spelling the column cannot read is not a second chance — `_eq: "1.0"` against a `UInt64` matches no rows on a read filter and is refused with `403` on an insert (an integer column takes only the canonical spelling; see the caution below), and on another numeric column such as `Decimal` it is ClickHouse's own code 53 `TYPE_MISMATCH` (72 on a `Float` column), which fails every `/v1/query` read by that role with `400 clickhouse.rejected`, withholds every row on the stream (reason `error`), and is a `422` on an insert. Write a literal the column can read. +Values are always bound as SQL **parameters**, never concatenated into the query, so templating is injection-safe. If a claim path in a `filter` template can't be resolved (a validly-signed token that simply doesn't carry the claim), that filter **fails closed**: the predicate becomes constant-false, so on the structured-query path (`POST /v1/query`) the role sees **no rows**, and the live stream withholds every event for that subscriber — one resolution drives both read surfaces (see [where each rule is enforced](#where-each-rule-is-enforced)). This holds for every operator — `_eq`, `_neq`, `_gt`, `_lt`, and `_in` alike. The alternative, binding the empty string the template would render to, would leave a live predicate against `''`: `_eq` would match every empty-valued row, and `_neq`/`_gt` on a string column would match essentially *all* rows, erasing the restriction. A literal value with no template in it — including an explicit `""` — binds exactly as written, and ClickHouse reads it under the column's type: a `String` column compares that exact text, while a numeric column reads the constant as a number. A spelling the column cannot read is not a second chance — `_eq: "1.0"` against a `UInt64` matches no rows on a read filter and is refused with `403` on an insert (an integer column takes only the canonical spelling; see the caution below), and a value another numeric column cannot read at all, such as `"abc"`, is ClickHouse's own code 53 `TYPE_MISMATCH` on a `Decimal` (72 on a `Float`), which fails every `/v1/query` read by that role with `400 clickhouse.rejected`, withholds every row on the stream (reason `error`), and is a `422` on an insert. Write a literal the column can read. "Can't be resolved" means the claim path is **absent** from the token (or `null`) — or resolves to a JSON **object or array** rather than a scalar, which usually means a dropped path segment (`{{ jwt.app_metadata }}` where `{{ jwt.app_metadata.tenant_id }}` was meant); the one structured shape with defined semantics is the bare-claim `_in` array above. Scalar claims — strings, booleans, and numbers — resolve normally. A numeric claim binds in **canonical decimal form**, not the token's spelling: `1.0` and `1e3` bind as `1` and `1000`, so spelling differences between issuers never change the bound value, and an integer id keeps every digit up to a ~100-digit bound on the value's *exact* decimal form (so `1e150` is refused even though a float64 holds it — prefer issuing integer ids as integers). Type the scoped column to match: an integer column, or a `Decimal` whose scale covers the claim's fractional digits, keeps the comparison exact for ids that fit the column's type, while a `Float32`/`Float64` column rounds the stored value and gives that exactness back (`col = '9007199254740993'` matches a stored `9007199254740992` there). A claim that is *present but empty* is a value the token vouches for: it resolves to `''` and binds normally. Make sure your identity provider issues the claims your policy templates reference — and omits unset claims rather than issuing them as empty strings. @@ -288,7 +288,7 @@ Checks are compiled into **one chtypes filter** — `col = {p:String}` for `_eq` - **If the request body includes the column**, its value must satisfy the check or the record is rejected with `403 check failed for column "x"` (`… for columns "x", "y"` when more than one is checked — the filter is AND-joined, so it names the set tested rather than inventing an attribution). Comparison is ClickHouse's, under the column's own type: on an integer, `Decimal`, `UUID` or `String` column a writer cannot forge a row for another tenant, because equal values store equal. A `Float32`/`Float64` column rounds the stored value and gives that exactness back — the row can land on a neighboring id. A check the engine cannot evaluate at all is a `422`, never a silent pass. - **If the request body omits the column** — or sends an explicit `null` on a non-`Nullable` column, which `input_format_null_as_default` resolves the same way — an `_eq` check **auto-injects** the claim-derived value, so clients can send just the business fields and let the policy stamp `user_id` and `tenant_id` from the token. On a `Nullable` column an explicit `null` is stored as `NULL`, which fails the check (`403`). It is implemented as a `DEFAULT` on the role's compiled schema, so a value the caller *does* send still wins. An `_eq` check also covers a column the role may not otherwise write (left out of `allow_columns`, or in `deny_columns`): a record that omits it is filled with the required value, one that supplies exactly that value is accepted, and any other value fails the check (`403`). An `_in` check has no single value to stamp, so the **table's own default** is what gets tested: the record is admitted if that default is in the claim-derived set and rejected if it is not. - **The check's column must be one a record can actually carry.** A `check` naming a column the table does not have, one ClickHouse computes (`MATERIALIZED`/`ALIAS`), or an `EPHEMERAL` one is refused with a per-record `403` naming the column — on *every* insert by that role, until the policy or the table is corrected. None of the three can be enforced: the published row has one slot per wire column, so an injected value for a computed or unknown column is dropped on the way out, and an ephemeral column is never stored. Each would have answered `200` while enforcing nothing. `wavehouse validate` cannot catch this — it never sees the ClickHouse schema — so **audit your `check` blocks against their tables before upgrading**. -- **A claim literal the column cannot read fails closed, not loosely.** A `_eq` value that is not a legal literal for the column (`"1.0"` on a `UInt64`) cannot be compiled as that column's `DEFAULT`, so the injection is dropped and logged; the filter then judges the record as sent, which an absent column loses. On an integer column every record then fails the check — a `403`; on another column the comparison itself errors on every record, supplied or omitted (ClickHouse's code 53 `TYPE_MISMATCH` on a `Decimal`, 72 on a `Float`) — a `422`. +- **A claim literal the column cannot read fails closed, not loosely.** A `_eq` value that is not a legal literal for the column (`"1.0"` on a `UInt64`, `"abc"` on a `Decimal` or a `Float`) cannot be compiled as that column's `DEFAULT`, so the injection is dropped and logged; the filter then judges the record as sent, which an absent column loses. On an integer column every record then fails the check — a `403`; on another column the comparison itself errors on every record, supplied or omitted (ClickHouse's code 53 `TYPE_MISMATCH` on a `Decimal`, 72 on a `Float`) — a `422`. **When the claim can't be resolved** (a validly-signed token that doesn't carry it), an `_eq` check is — unlike a row filter — **not** fail-closed on a `String` column: the template still renders, the unresolvable placeholder replaced by the empty string and any surrounding literal text kept (`"acct-{{ jwt.org_id }}"` → `acct-`, a bare `"{{ jwt.sub }}"` → `''`), and that rendered value becomes the required value — so an omitted column is auto-injected with it and any other supplied value is rejected. On a column whose type cannot read `''` — a number, a `Date`/`DateTime`, a `UUID` — it now fails closed: `''` is not a literal a `UInt64` can read, so the role's schema will not compile with it as a default and no record satisfies the check — every insert by that role into that table is refused (a `403` on an integer column, a `422` on any other), rather than the rows landing on tenant/user `0` as they used to ([#463](https://github.com/Wave-RF/WaveHouse/issues/463)). An unresolvable `_in` check fails closed everywhere: the claim-derived set is empty, so nothing satisfies it. The `_in` element rule from [row-level security](#row-level-security) applies here too — one non-scalar element (`null`, object, nested array) in the claim array collapses the allowed set the same way. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 078fa2f5..4efe625a 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -264,7 +264,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | Status | Body | Cause | | ------ | ---- | ----- | -| 400 | `{"error":"","exception_code":}` | **Per-record.** ClickHouse's parser refused the record; `exception_code` and the message are its own. `117` is an unknown field — which now includes a column the role may not write and any `MATERIALIZED`/`ALIAS` column; `27`/`26` are input it cannot parse and `33` a record cut off mid-object; other codes are type-specific (`41` a `DateTime`/`DateTime64`, `38` a `Date`, `69` a `Decimal` with too many digits, `72` a number it cannot read — no digits, or a negative into a `UInt*`, `376` a `UUID`, `467` a `Bool`, `691` an unknown `Enum` element, `675` an `IPv4`). An out-of-range integer is not rejected: it wraps (above) | +| 400 | `{"error":"","exception_code":}` | **Per-record.** ClickHouse's parser refused the record; `exception_code` and the message are its own. `117` is an unknown field in a name-addressed format — which now includes a column the role may not write and any `MATERIALIZED`/`ALIAS` column — or too many fields on a positional CSV or TSV line; `27`/`26` are input it cannot parse and `33` a record cut off mid-object; other codes are type-specific (`41` a `DateTime`/`DateTime64`, `38` a `Date`, `69` a `Decimal` with too many digits, `72` a number it cannot read — no digits, or a negative into a `UInt*`, `376` a `UUID`, `467` a `Bool`, `691` an unknown `Enum` element, `675` an `IPv4`). An out-of-range integer is not rejected: it wraps (above) | | 400 | `{"error":"Unknown field found in format header: 'x' at position 1 …","code":"clickhouse.rejected","exception_code":117}` | A `header=present` body whose header names a column the table — or the role's writable set — does not have, or names one twice. ClickHouse refuses the body before reading any record, so the whole request fails and nothing is published | | 400 | `{"error":"missing table"}` | No `table` query parameter. Checked before the body is read | | 400 | `{"error":"invalid request body"}` | The body could not be read at all — a malformed transfer encoding, or a truncated upload (a body cut off *in transit*). A body that arrived complete but ends mid-value is not this error: a JSON array cut short is `invalid json: unterminated json array` below, while a single object or NDJSON cut mid-value is a per-record ClickHouse rejection (code 26, 27 or 33) — or, cut right after a key's `:` or an opening quote, a `422` decline of every record in the body | @@ -291,7 +291,7 @@ WaveHouse decides only policy: whether the role may insert at all, and whether t | 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure, for that tenant alone) or not open (see [Message Queue](/settings-directory#message-queue)). Under [`mq.backend: nats`](/deployment#external-nats), the table's partition stream is full, which refuses every table in it, or the tenant's table holds as many unwritten rows as the stream allows one subject. Response includes `Retry-After: 30` header. With dedupe on, the record's id is given back, so the retry is published rather than reported as a duplicate. | | 503 | `{"error":"service unavailable"}` | The message queue could not be reached or did not answer in time (`mq.ErrUnavailable`, a transient broker failure, not a full queue). Only under [`mq.backend: nats`](/deployment#external-nats), including a partition stream the operator deleted; the embedded broker never reports this, and its publish failures are the `500` above. As for the `500`, the record's id is left to lapse rather than given back, so a retry cannot land as a second copy; `Retry-After` is that lease, rounded up to whole seconds, when dedupe was on for the record, else the flat `Retry-After: 5`. | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | Dedupe is on and another request carrying the same id is still being published — usually a client's timeout-retry racing its own original. Its outcome decides whether this record is a duplicate, so retry after the `Retry-After` header (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). | -| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet (no completed schema refresh), its ClickHouse line has no installed chtypes artifact for its server version, its server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), or the table's schema could not be compiled. The body is generic on purpose — the cause, with zone names and artifact search paths, goes to the server log only. Only that tenant (or, for a table that could not be compiled, only that table) is refused — every other tenant keeps working — and it is decided before the body is read, with `Retry-After: 5`. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for how each cause recovers | +| 503 | `{"error":"ingest validation is unavailable"}` | The tenant is not bound in the type layer (only for a request racing a reload that stops serving it: before its first completed schema refresh the answer is `schema not loaded yet`, above), its ClickHouse line has no installed chtypes artifact for its server version, its server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line), or the table's schema could not be compiled. The body is generic on purpose — the cause, with zone names and artifact search paths, goes to the server log only. Only that tenant (or, for a table that could not be compiled, only that table) is refused — every other tenant keeps working — and it is decided before the body is read, with `Retry-After: 5`. See [Deployment → chtypes artifacts](/deployment#chtypes-artifacts) for how each cause recovers | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | **curl example:** @@ -430,7 +430,7 @@ A `200` is returned whenever the body was read and the records were processed | 503 | `{"error":"dedupe store unavailable"}` | Dedupe is on and its store cannot answer now; `Retry-After: 5`. Nothing in the window being reserved was published; the windows before it were, and keep their ids | | 503 | `{"error":"a request with the same dedupe id is in flight"}` | A record's dedupe id is held by another request still being published; includes `Retry-After` (the dedupe lease, [`dedupe.lease`](/configuration#dedupe), 30 seconds by default). Nothing in that record's window was published; the windows before it were | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet; `Retry-After: 5`. Decided before the body is read | -| 503 | `{"error":"ingest validation is unavailable"}` | The tenant's schema is not bound yet, its ClickHouse line has no installed chtypes artifact, its server time zone differs from the zone this process opened that line with, or the table's schema could not be compiled (that table alone) — the cause is in the server log, not the body. Decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | +| 503 | `{"error":"ingest validation is unavailable"}` | The tenant is not bound in the type layer (a request racing a reload that stops serving it; before its first refresh the answer is `schema not loaded yet`), its ClickHouse line has no installed chtypes artifact, its server time zone differs from the zone this process opened that line with, or the table's schema could not be compiled (that table alone) — the cause is in the server log, not the body. Decided once, before the body is read, so nothing is published. `Retry-After: 5`. See the [single-record table](#post-v1ingesttabletable--ingest-data) | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] @@ -689,7 +689,7 @@ id: 2026-03-24T12:00:01.456Z data: {"table_name":"clicks","received_timestamp":"2026-03-24T12:00:01.456Z","row":["/pricing","cta",7,"2026-03-24T12:00:01.456Z"]} ``` -A `/` in a string arrives as `/`, not as the `\/` ClickHouse's JSON writer produces by default: WaveHouse pins `output_format_json_escape_forward_slashes=0` on the export that produces these rows and on `/v1/query`, pipes and `/v1/ops/query`. +A `/` in a string arrives as `/`, not as the `\/` ClickHouse's JSON writer produces by default: WaveHouse pins `output_format_json_escape_forward_slashes=0` on the export that produces these rows and on `/v1/query`, pipes and `/v1/ops/query`. A `Float*` NaN or infinity arrives as a string (`"nan"`, `"inf"`, `"-inf"`), where `/v1/query` and pipes render `null`. The stream forwards the published row's cells as they are, and that row is what the ingest worker inserts: the string is what stores the NaN or infinity (a `null` would store the column's default), and the export that produces it does not read `output_format_json_quote_denormals`. A raw consumer must keep the most recent announced column list and zip each `row` against it; a column the record omitted still has its slot, holding its evaluated `DEFAULT` (or the type's default — `null` only on a `Nullable` column with none), so positions never shift. **Check arity before zipping:** drop a `row` whose length disagrees with the last announced list rather than zipping it, because the announcement is not guaranteed in one case — a connection that gap-fills across a column change may receive live rows with no fresh announcement until the columns next change or it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). An arity check covers an added or removed column; a *same-length* change (a `RENAME COLUMN`, or a drop paired with an add) it cannot see, and reconnecting is what resynchronizes. Separately, a replay spanning a server upgrade across the v2 ingest envelope silently omits the pre-upgrade events — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope). The TypeScript SDK does this for you: `.stream()` and `.liveQuery()` zip each row into an object. The announcement is **per connection**, so a client that joins mid-stream is told the columns before it is sent a row, and a reconnect is told again. diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index ff88a41e..28de54ea 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -116,15 +116,15 @@ gh attestation verify oci://ghcr.io/wave-rf/wavehouse:vX.Y.Z \ **Which processes load it.** Only processes with the `api` role — the ones that serve ingest and the stream. A process that runs only the ingest worker or the sweeper loads no artifact and boots without one installed. An API process refuses to start when no artifact is installed at all. -**When a tenant or table is unavailable.** The type layer refuses a whole tenant, on its own, for three causes: the tenant is not bound yet (its schema discovery has not completed a refresh); its ClickHouse line has no installed artifact (or only one the SDK refuses to load, such as a truncated library); or its server reports a different time zone from the one this process already opened that line with — the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**, and the first tenant bound on a line fixes that zone. A fourth cause refuses one table and leaves the tenant's other tables working: a table whose schema the parser could not compile. In every case, ingest into that tenant (or that table) answers `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable`, decided before the body is read; a stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable`, while roles with no row filter are unaffected; and every other tenant keeps working. +**When a tenant or table is unavailable.** The type layer refuses a whole tenant, on its own, for three causes: the tenant is not bound yet (its schema discovery has not completed a refresh); its ClickHouse line has no installed artifact (or only one the SDK refuses to load, such as a truncated library); or its server reports a different time zone from the one this process already opened that line with — the library reads its time zone once, when a line is first opened, so one process serves **one server time zone per ClickHouse line**, and the first tenant bound on a line fixes that zone. A fourth cause refuses one table and leaves the tenant's other tables working: a table whose schema the parser could not compile. In every case a stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable`, while roles with no row filter are unaffected, and every other tenant keeps working. Ingest into that tenant (or that table) answers `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable`, decided before the body is read — except for a tenant not bound yet, whose schema is not loaded yet either: the type layer is bound in the same refresh that marks the schema loaded, so its ingest answers `503` `schema not loaded yet`, and only a request racing a reload that stops serving the tenant meets the type layer's `503`. -**How it recovers, and what to watch.** Every cause is checked again at each of the tenant's schema refreshes (`schema.refresh_interval`), and clears at the first refresh after the cause is gone. A tenant not bound yet clears at its first completed refresh. A missing artifact clears once a loadable one is installed on the search path, with no restart. A time zone mismatch clears once the server reports the zone that line was opened in, so align the servers' `timezone` setting; otherwise serve such tenants from a separate process. A table that does not compile is compiled again at every refresh, and clears once its schema, or the artifact answering for it, changes so that it compiles. The health probes stay green through all four — `/livez` and `/readyz` do not read the type layer — so watch for the ingest `503`s, for `wavehouse_sse_rows_withheld_total{reason="unavailable"}`, and for three `ERROR` log lines, each carrying the cause: `chtypes cannot serve this tenant` (a missing artifact or a time zone mismatch, logged at every refresh), `chtypes could not compile table schema` (one table, logged at every refresh) and `ingest type layer unavailable` (any of the four, logged for the ingest requests it refuses, at most once a minute per tenant and table). See [API → Ingest error responses](/api#post-v1ingesttabletable--ingest-data) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). +**How it recovers, and what to watch.** Every cause is checked again at each of the tenant's schema refreshes (`schema.refresh_interval`), and clears at the first refresh after the cause is gone. A tenant not bound yet clears at its first completed refresh. A missing artifact clears once a loadable one is installed on the search path, with no restart. A time zone mismatch clears once the server reports the zone that line was opened in, so align the servers' `timezone` setting; otherwise serve such tenants from a separate process. A table that does not compile is compiled again at every refresh, and clears once its schema, or the artifact answering for it, changes so that it compiles. `/livez` and `/readyz` do not read the type layer, so they stay green through the last three causes; a tenant not bound yet has not completed a schema discovery, which they report as `503` only while no tenant has completed one ([Boot-time degraded mode](#boot-time-degraded-mode)). So watch for the ingest `503`s, for `wavehouse_sse_rows_withheld_total{reason="unavailable"}`, and for three `ERROR` log lines, each carrying the cause: `chtypes cannot serve this tenant` (a missing artifact or a time zone mismatch, logged at every refresh), `chtypes could not compile table schema` (one table, logged at every refresh) and `ingest type layer unavailable` (any of the four, logged for the ingest requests it refuses, at most once a minute per tenant and table). See [API → Ingest error responses](/api#post-v1ingesttabletable--ingest-data) and [Access Control → Where each rule is enforced](/access-control#where-each-rule-is-enforced). **Where it lives.** WaveHouse looks for the artifact in a registry directory, in order: an explicit `clickhouse.chtypes_registry` (`WH_CHTYPES_REGISTRY`) if set, then chtypes' own default search path — `$CHTYPES_REGISTRY`, the per-user cache `~/.cache/chtypes/artifacts/abi6/-` (one directory per SDK ABI revision, so an older SDK's downloads are never picked up), then the system directories `/usr/local/share/chtypes/artifacts/` and `/opt/chtypes/artifacts/`. WaveHouse never fetches an artifact itself: a line with no installed artifact makes the tenants on it unavailable at their schema refresh, and boot is refused only when no artifact is installed at all, or when an explicit `clickhouse.chtypes_registry` does not exist or cannot be read. One exception: the SDK honors `CHTYPES_AUTOFETCH=1` from the environment over WaveHouse's setting, and would then download a missing line (hundreds of MB) inside a schema refresh — leave it unset. **Size.** Each artifact is roughly 160–300 MB on disk; a running process holding several loaded versions (e.g. across a rolling ClickHouse upgrade) costs roughly 120 MB of resident memory per loaded version (the chtypes multi-version guide's figure; a library is opened on first use of its line, not at registry construction). On top of the library, every table of every tenant an API process serves holds compiled parser handles, roughly 40 KiB each and 96 KiB once warm: one from the moment its schema is bound, more only while every handle is busy, up to `GOMAXPROCS` and at most 8, plus up to 256 role shapes per table (one per distinct set of writable columns and claim-stamped defaults among the roles that insert) of up to 4 handles each. Each handle also caches up to 4,096 compiled filters, one per distinct expression and claim values, row filters and insert checks alike: about 10 KiB for a one-clause filter and 27 KiB for two clauses with a three-value `in` (measured on the 26.8 artifact, darwin arm64), so a full cache of one-clause filters is about 43 MiB, and a table's 8 base handles about 340 MiB. A cache fills only once that many distinct claim sets have been evaluated on one handle — a row-filtered table streamed to thousands of subscribers whose claims differ — and then stays full, evicting the least recently used. All of it multiplies by tables and by tenants. -**Docker images** ship the artifact(s) baked in: the image build fetches the lines `scripts/fetch-chtypes.sh` lists (`LOCK_LINES`), each pinned by `chtypes.lock` (see below), so a container never needs network access to chtypes' artifact store at runtime. The image sets `CHTYPES_REGISTRY=/opt/chtypes/artifacts` (the SDK's own variable); set `WH_CHTYPES_REGISTRY` only to point at a bind-mounted directory instead. **The published images support ClickHouse 26.8 only**: they bake only that line, and a server on any other line answers every tenant on it `503` (with row-filtered streams withholding their rows) behind a generic body. For another line, fetch it for the container's platform into a directory of its own — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --platform linux-amd64 --dest ./chtypes-artifacts ` (`linux-arm64` on arm) — make it readable by the image's user (`chmod -R a+rX ./chtypes-artifacts`; the fetch writes each line's directory with mode `0700`), mount it read-only, and set `WH_CHTYPES_REGISTRY` to the mount path. That directory is searched first and the baked line stays available; a path that does not exist or cannot be read refuses boot. +**Docker images** ship the artifact(s) baked in: the image build fetches the lines `scripts/fetch-chtypes.sh` lists (`LOCK_LINES`), each pinned by `chtypes.lock` (see below), so a container never needs network access to chtypes' artifact store at runtime. The image sets `CHTYPES_REGISTRY=/opt/chtypes/artifacts` (the SDK's own variable); set `WH_CHTYPES_REGISTRY` only to point at a bind-mounted directory instead. **The published images support ClickHouse 26.8 only**: they bake only that line, and ingest for every tenant on a server of any other line answers `503` (and its row-filtered streams withhold their rows) behind a generic body. For another line, fetch it for the container's platform into a directory of its own — `go run github.com/wave-rf/chtypes/go/cmd/chtypes@v0.5.2 fetch --platform linux-amd64 --dest ./chtypes-artifacts ` (`linux-arm64` on arm) — make it readable by the image's user (`chmod -R a+rX ./chtypes-artifacts`; the fetch writes each line's directory with mode `0700`), mount it read-only, and set `WH_CHTYPES_REGISTRY` to the mount path. That directory is searched first and the baked line stays available; a path that does not exist or cannot be read refuses boot. **Release archives and `go install` / building from source** do not carry or fetch an artifact — only the Docker images bake one in. See the [README's `go install` caveat](https://github.com/Wave-RF/WaveHouse#c-go-install-binary-no-docker). Fetch one yourself before first run. Both routes below run the chtypes command with `go run`, which needs Go 1.27 and a C compiler; a release archive has neither `scripts/fetch-chtypes.sh` nor the lock file, so use the second form there, or fetch from a checkout and copy the directory to the host: @@ -615,7 +615,7 @@ The folder name is the tenant id, and each folder is a complete settings directo **The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` takes it too, and reads a rejected or removed tenant's dead-letter queue like a served one's, since the queue is kept; under the embedded broker a tenant that has none is a `404`. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). -**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant and table), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it, and in each table. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Tenants on one ClickHouse line must share a server time zone to be served by one process: the first tenant bound fixes the line's zone, and a tenant whose server reports another is refused (`503`; see [chtypes artifacts](#chtypes-artifacts)). Under the embedded broker its message queue is its own as well (under [`mq.backend: nats`](#limits-that-differ-from-the-embedded-queue) tenants share the partitions): its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history that gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a rejected row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached results (`POST /v1/query` and pipe alike) dropped the moment it is back on one, so a repaired or restored folder never serves rows cached before the inserts it missed. A tenant whose folder moves it to another address or database has them dropped too, since they came from other tables. Apart from those two drops, a cached pipe result stays until its TTL expires, since a pipe names no table and no insert invalidates it. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. +**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant and table), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it, and in each table. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Tenants on one ClickHouse line must share a server time zone to be ingested into and row-filtered by one process: the first tenant bound fixes the line's zone, and a tenant whose server reports another has its ingest refused (`503`) and its row-filtered streams withheld (see [chtypes artifacts](#chtypes-artifacts)). Under the embedded broker its message queue is its own as well (under [`mq.backend: nats`](#limits-that-differ-from-the-embedded-queue) tenants share the partitions): its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history that gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a rejected row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached results (`POST /v1/query` and pipe alike) dropped the moment it is back on one, so a repaired or restored folder never serves rows cached before the inserts it missed. A tenant whose folder moves it to another address or database has them dropped too, since they came from other tables. Apart from those two drops, a cached pipe result stays until its TTL expires, since a pipe names no table and no insert invalidates it. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. **What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. Tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its seen ids kept for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its open `GET /v1/stream` connections are ended, as any tenant's are when it stops being served — the other tenants' events are untouched. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. @@ -651,7 +651,7 @@ For local development, `docker compose -f deployments/compose/dependencies.yaml WaveHouse uses a **Bring Your Own Schema** model. You create your tables in ClickHouse with whatever columns and engines you need. WaveHouse discovers the schemas automatically via `system.columns` and validates ingest data against them — see [Schema Validation](/api#post-v1ingesttabletable--ingest-data) for the rules a record must satisfy. -Five schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's default where none is declared (`NULL` on a `Nullable` column), evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [Ingest Pipeline → High-level shape](/ingest-pipeline#high-level-shape) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). +Four schema-design consequences are worth knowing before you write the DDL. A `MATERIALIZED`, `ALIAS`, or `EPHEMERAL` column is never part of a published row: WaveHouse's ingest validation runs ClickHouse's own parser in-process (via [chtypes](#chtypes-artifacts)), and a record that names a `MATERIALIZED` or `ALIAS` one is rejected with ClickHouse's own code (117) rather than published, while an `EPHEMERAL` value is accepted only where the format names columns (the JSON family, `…WithNames`), the role may write it and a `DEFAULT` reads it (and no `MATERIALIZED`, `ALIAS` or other `EPHEMERAL` column does), and then feeds that `DEFAULT` without being stored or published (anywhere else it is code 117); a policy `check` naming any of the three is refused outright. An omitted column — on any table — takes its `DEFAULT` expression, or the type's default where none is declared (`NULL` on a `Nullable` column), evaluated by that same parser before the row is published; there is no longer a positional-encoding quirk that stores `NULL` on a `Nullable(T) DEFAULT …` column instead — see [Ingest Pipeline → High-level shape](/ingest-pipeline#high-level-shape) for detail. Rows retried after a ClickHouse outage reach ClickHouse out of ingest order, so a table whose engine picks a winner by insert order — a `ReplacingMergeTree` without a version column, a `CollapsingMergeTree` — needs a version column the producer sets in the record (`ReplacingMergeTree(ver)`, `VersionedCollapsingMergeTree`), not an insert-time `DEFAULT now64()` like the example's `received_timestamp`. And a retry after an insert whose outcome WaveHouse could not see (a timeout, a dropped connection) can land its rows twice on any engine — the example's plain `MergeTree` included, and a `VersionedCollapsingMergeTree` then keeps a state row its one cancel cannot remove — so a table that must not count a row twice needs a `ReplacingMergeTree` keyed on an id the producer sets, read with `FINAL` (it removes a duplicate only when parts merge; a [pipe](/pipes) can say `FINAL`, a structured query never adds it), or reads that tolerate duplicates, such as `uniqExact(id)`. `dedupe.enabled` does not prevent this: it drops a repeated publish at the HTTP edge, and this duplicate is made after the queue. See [When ClickHouse cannot take an insert](/ingest-pipeline#when-clickhouse-cannot-take-an-insert). Example table: diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index f9545fef..f6d84f95 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -13,7 +13,7 @@ You need these on your `PATH` before any `make` recipe will work end-to-end: | Tool | Required version | Why | Install | | ---- | ---------------- | --- | ------- | -| **Go** | 1.27+ (matches `go.mod`) | Compiles `cmd/wavehouse` with cgo enabled (needed by chtypes' dlopen shim — a C toolchain, and on Linux glibc 2.34 or later, must be present); also runs the pinned `tool` deps (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `deadcode`, `gsa`, `goda`) via `go tool` | [go.dev/dl](https://go.dev/dl/) | +| **Go** | 1.27+ (matches `go.mod`) | Compiles `cmd/wavehouse` with cgo enabled (needed by chtypes' dlopen shim — a C toolchain must be present; on Linux the binary links against the build host's glibc, where the prebuilt release binaries need 2.34 or later); also runs the pinned `tool` deps (`gotestsum`, `gofumpt`, `goimports`, `govulncheck`, `deadcode`, `gsa`, `goda`) via `go tool` | [go.dev/dl](https://go.dev/dl/) | | **GNU Make** | **4.0+** | The Makefile uses `--output-sync=target` (Make 4 only) and bash-pinned recipes. macOS ships with BSD Make 3.81, which **will not work** | macOS: `brew install make` then use `gmake` or put `$(brew --prefix make)/libexec/gnubin` on your PATH. Linux: usually already installed | | **bash** | 4+ recommended | Recipes are pinned to `bash`; the helper scripts under `scripts/` use `set -euo pipefail` and bash arrays | macOS default is bash 3.2 (works for current recipes, but `brew install bash` is safer); Linux distros ship 4+ | | **Docker** *(or Podman)* | Engine 20.10+ with the Compose **v2** plugin (`docker compose`, no hyphen) | Compose stacks under `deployments/compose/`; the E2E and integration suites boot ClickHouse and a Redis via testcontainers (no compose file), the integration suite also dynamodb-local, and the integration suite also runs the shared cache backend against Redis, Valkey, Dragonfly (pulled from `docker.dragonflydb.io`) and a one-node Redis Cluster | [Docker Desktop](https://docs.docker.com/get-docker/), [colima](https://github.com/abiosoft/colima), or [Podman](https://podman.io) with `podman-compose` / the `podman compose` plugin. The testcontainers Go library also honors `DOCKER_HOST` for rootless Podman setups | @@ -71,7 +71,7 @@ cd WaveHouse make tools scripts/fetch-chtypes.sh # once per machine (see above) -# 2. Start ClickHouse (the only external dependency) +# 2. Start ClickHouse (the only external service) docker compose -f deployments/compose/dependencies.yaml up -d --wait clickhouse # 3. Create a table in ClickHouse diff --git a/docs/src/content/docs/getting-started.md b/docs/src/content/docs/getting-started.md index 1e231737..5d7ace66 100644 --- a/docs/src/content/docs/getting-started.md +++ b/docs/src/content/docs/getting-started.md @@ -11,7 +11,7 @@ Run WaveHouse locally in under five minutes. WaveHouse ships as one binary plus - **Docker** — for running ClickHouse (and optionally WaveHouse itself). - **curl** and **jq** (optional) — for poking the API. -- **Go 1.27+** — only required if you want to build from source; skip it for the Docker path below. Building from source also requires cgo (a C toolchain; on Linux, glibc 2.34 or later) — see [Deployment → Supported Platforms](/deployment#supported-platforms). +- **Go 1.27+** — only required if you want to build from source; skip it for the Docker path below. Building from source also requires cgo (a C toolchain; on Linux the binary links against the build host's glibc, where the prebuilt binaries need 2.34 or later) — see [Deployment → Supported Platforms](/deployment#supported-platforms). ## 1. Start WaveHouse diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index 1e9cbab2..8ed8cbce 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -173,7 +173,7 @@ export interface ClicksRow { | ClickHouse Type | TypeScript Type | |----------------|-----------------| | `String`, `FixedString`, `UUID`, `DateTime*`, `Date*`, `Enum*`, `IPv4/6` | `string` | -| `UInt*`, `Int*`, `Float*`, `Decimal*` | `number` — 64-bit and wider integers and `Decimal*` come back as JSON numbers, not strings, so `JSON.parse` rounds a value past 2^53 — store an id that large as a `String` column to keep every digit; a `Float*` NaN or infinity comes back as `null` | +| `UInt*`, `Int*`, `Float*`, `Decimal*` | `number` — 64-bit and wider integers and `Decimal*` come back as JSON numbers, not strings, so `JSON.parse` rounds a value past 2^53 — store an id that large as a `String` column to keep every digit; a `Float*` NaN or infinity comes back as `null` from queries and pipes, and as a string (`"nan"`, `"inf"`, `"-inf"`) on a stream | | `Bool` | `boolean` | | `Nullable(T)` | `T \| null` | | `Array(T)` | `T[]` | diff --git a/internal/typelayer/ingest_test.go b/internal/typelayer/ingest_test.go index 1887a98f..200c901c 100644 --- a/internal/typelayer/ingest_test.go +++ b/internal/typelayer/ingest_test.go @@ -320,6 +320,30 @@ func TestIngest_ForwardSlashExportsUnescaped(t *testing.T) { } } +// TestIngest_NaNAndInfinityExportAsStrings: the export spells a Float NaN or +// infinity as a JSON string, the spelling the worker's INSERT stores as that +// value (a null would store the column's default), so the stream carries +// strings where /v1/query renders null. The export does not read +// output_format_json_quote_denormals, so it cannot be pinned to agree. +func TestIngest_NaNAndInfinityExportAsStrings(t *testing.T) { + eng := testEngine(t, &discovery.TableSchema{Name: "floats", Columns: []discovery.Column{ + {Name: "f", Type: "Float64", Position: 1}, + {Name: "g", Type: "Nullable(Float32)", Position: 2}, + }}) + tbl, err := eng.Table(tenant.Default, "floats") + require.NoError(t, err) + t.Cleanup(tbl.Release) + + batch, err := tbl.Ingest(FormatJSONEachRow, []byte(`{"f":"nan","g":"inf"}`+"\n"+`{"f":"-inf","g":null}`+"\n")) + require.NoError(t, err) + require.Len(t, batch.Rows, 2) + for _, r := range batch.Rows { + require.True(t, r.Accepted, r.Message) + } + assert.Equal(t, `["nan", "inf"]`, string(batch.Rows[0].Line)) + assert.Equal(t, `["-inf", null]`, string(batch.Rows[1].Line)) +} + func TestInsertSettings_ReturnsAFreshMap(t *testing.T) { t.Parallel() a := InsertSettings() From a3188307a59524e9a39cbfc97a0aef09ec589a3f Mon Sep 17 00:00:00 2001 From: Eric Andrechek Date: Thu, 1 Oct 2026 14:30:55 -0400 Subject: [PATCH 70/70] fix(ingest): read BOM- and form-feed-led bodies the way ClickHouse does The array-or-object byte skipped only space, tab, CR and LF, so a JSON array led by a UTF-8 byte order mark, a form feed or a vertical tab and declared application/json took the single-object path: it was never re-framed and only element 0 was published, behind 200 {"ok":true}. ClickHouse's JSON reader skips the mark at the very start and all six ASCII whitespace bytes, so firstNonSpace and reframeArray now do too and the array gets the batch response with every element published. A body that is only a mark and whitespace is the empty-body 400. The CSV/TSV record floor declined bodies chtypes read whole. Measured on 26.8.15.10, ClickHouse skips a leading mark under a WithNames header and before a non-string first column, so `BOM id,page,n` is a detected header on a UUID- or DateTime-first table, while a String or FixedString first column (Nullable/LowCardinality too) keeps it as the value. The floor now counts a marked body that way, and takes the lower of the two readings for a first column not known to keep it, so a marked header a String-first table reads as a record still declines when it swallows the next line. A CSV LF CR is one line end to ClickHouse, like CRLF, and is counted as one; a TSV CR after LF opens the next record (measured) and stays counted that way. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- docs/src/content/docs/api.md | 4 +- docs/src/content/docs/architecture.md | 2 +- internal/api/content_type.go | 35 ++++++++--- internal/api/ingest_count_test.go | 87 +++++++++++++++++++++++++- internal/api/ingest_framing.go | 17 ++++-- internal/api/ingest_framing_test.go | 40 ++++++++++++ internal/typelayer/ingest.go | 6 +- internal/typelayer/records.go | 78 +++++++++++++++++++++--- internal/typelayer/records_test.go | 88 +++++++++++++++++++++++++-- 10 files changed, 325 insertions(+), 34 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index b3d6c010..ca4c3dd6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -63,7 +63,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **The release pipeline builds each binary on its own native runner; GoReleaser is now only the compiler** (`.goreleaser.yaml`, `.github/workflows/release.yml`, `.github/workflows/publish-dev.yml`, `.github/workflows/goreleaser-validate.yml`, `deployments/Dockerfile.goreleaser`, `docs/src/content/docs/development.md`): cgo cannot cross-compile darwin from Linux. Measured rather than inferred — `zig cc -target aarch64-macos` fails at *compile* time on `prometheus/client_golang`'s `process_collector_mem_cgo_darwin.c`, which `#include`s ``; `-tags netgo,osusergo` does not help, because the build never reaches the linker that the earlier `-lresolv` finding was about, and no Apple SDK can be fetched onto a GitHub-hosted Linux runner. GoReleaser's answers to this (split/merge, `builder: prebuilt`) are Pro-only and OSS `goreleaser release` accepts no `--skip=build`, so it cannot assemble a release from binaries built elsewhere. `release.yml` therefore runs `goreleaser build --single-target` on `ubuntu-latest`, `ubuntu-24.04-arm` and `macos-latest` — all free for public repos — and one `ubuntu-latest` job assembles the `.tar.gz` archives, `checksums.txt`, the multi-arch GHCR image (`docker buildx build` over the unchanged `Dockerfile.goreleaser`, given the same `//wavehouse` context layout `dockers_v2` used to produce), the GitHub Release and both provenance attestations. `.goreleaser.yaml` shrinks to `project_name`, `git.ignore_tags` and `builds:`, and keeps being the one declaration of the ldflags, binary name and supported platform set; its per-target `CC`/`CXX` overrides are gone. Behaviour is preserved deliberately, not incidentally: archive names and contents, `checksums.txt` format, the immutable-tag-plus-channel-pointer scheme via `scripts/ci/release-channel.sh`, `prerelease: auto` (now "the channel is not `latest`"), `mode: keep-existing` (now a `gh release view` guard, which also makes the job re-runnable) and `changelog.use: github-native` with `git.ignore_tags` (now `gh release create --generate-notes` with `--notes-start-tag` set to the previous `v*` tag, `git describe --tags --abbrev=0 --match 'v*' "${tag}^"` — without that flag GitHub would happily diff a server release against a `clients/ts/v*` one). `publish-dev.yml` follows the same shape with only the two Linux targets, since a dev build publishes only the image. `goreleaser-validate.yml` becomes a real proof instead of a host-platform-only smoke test: `goreleaser check`, all three targets in `--snapshot`, and a genuine multi-arch `docker buildx build` to `--output type=cacheonly`, which also exercises the `chtypes.lock` fetch — the one PR-time signal that would have caught an upstream artifact republish before a tag did. Note the released **Linux binaries are now dynamically linked and require `GLIBC_2.34`** (measured on `ubuntu-24.04`, both architectures: Debian 12 / Ubuntu 22.04 / RHEL 9 and newer); the pre-cgo builds were static. Container images are unaffected — `distroless/cc-debian12` is glibc 2.36. -- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser in one call per request, as-is except that a JSON array is re-framed in place to one element per line so one bad element cannot lose the rest, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable (`33` for a record cut off mid-object), `117` unknown field, otherwise the type's own code (`72`, `41`, `38`, `69`, `376`, `467`, `691`, `675` …); an out-of-range integer wraps exactly as ClickHouse's own `INSERT` does (`256` → `0` in a `UInt8`) — so a malformed record is no longer the whole-request `400 {"error":"invalid json"}` (a JSON array whose brackets do not balance, or with content after its `]`, is still a whole-request `400 {"error":"invalid json: …"}`, and so is one with an empty element — a leading, doubled or trailing comma); a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; a body chtypes answers fewer records for than WaveHouse counts in it — a `UUID` shorter than 36 characters makes ClickHouse's reader consume the records after it, which then get no verdict while the batch is still accepted — is declined whole, every counted record `422` and nothing published, rather than reported as a smaller batch; and **timestamp values on the wire — published rows, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date` and `Date32` keep their `"2026-06-21"` form on the stream, and on `/v1/query` and pipes they lose the `T00:00:00Z` the native path added — see the structured-query entry) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant not bound yet, one whose ClickHouse line has no installed artifact, or one whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line) is refused on its own, and so is a table whose schema chtypes could not compile (that table alone) — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable` — while every other tenant keeps working. Each cause is checked again at every schema refresh and clears at the first one after it is fixed ([Deployment → chtypes artifacts](https://wavehouse.dev/deployment#chtypes-artifacts)). Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A dedupe id is read from the exported row: a `null` cell or an empty string is missing (so is an omitted `String` id with no `DEFAULT`), and a numeric id column cannot tell an omitted `0` from a supplied one. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. +- **The type layer is ClickHouse's own: ingest validation, row-level security and insert checks all run through chtypes** (BREAKING; new `internal/typelayer` package wrapping `github.com/wave-rf/chtypes/go` v0.5.2, a cgo dlopen of a per-ClickHouse-version shared library, loaded only by a process running the `api` role; `internal/discovery`, `internal/api/{ingest,content_type,ingest_framing}.go`, `internal/ingest/worker.go`, `internal/stream/{hub,roweval}.go`, `internal/policy`): the hand-written type-coercion, validation and row-filter code is replaced by calls into the same parser/analyzer ClickHouse's own server runs, loaded per ClickHouse minor line rather than compiled in. **The request body is no longer decoded in Go at all** — it goes to that parser in one call per request, as-is except that a JSON array is re-framed in place to one element per line so one bad element cannot lose the rest, and what comes back is a verdict per record plus the accepted rows as the exact `JSONCompactEachRow` bytes ClickHouse's writer produced. Consequences, all BREAKING: per-record errors carry ClickHouse's own message and its numeric code as `exception_code` (`{"exception_code": , "error": ""}`, with no string `code`; a whole-request parser refusal is `code: "clickhouse.rejected"` plus `exception_code`) — `27`/`26` unparseable (`33` for a record cut off mid-object), `117` unknown field, otherwise the type's own code (`72`, `41`, `38`, `69`, `376`, `467`, `691`, `675` …); an out-of-range integer wraps exactly as ClickHouse's own `INSERT` does (`256` → `0` in a `UInt8`) — so a malformed record is no longer the whole-request `400 {"error":"invalid json"}` (a JSON array whose brackets do not balance, or with content after its `]`, is still a whole-request `400 {"error":"invalid json: …"}`, and so is one with an empty element — a leading, doubled or trailing comma); a JSON body led by a UTF-8 byte order mark, a form feed or a vertical tab is read the way ClickHouse reads it, an array as an array, where 0.1.0 refused it; a record the engine cannot answer for is `422 "validation engine declined: …"`, never a `400`; a body chtypes answers fewer records for than WaveHouse counts in it — a `UUID` shorter than 36 characters makes ClickHouse's reader consume the records after it, which then get no verdict while the batch is still accepted — is declined whole, every counted record `422` and nothing published, rather than reported as a smaller batch; and **timestamp values on the wire — published rows, SSE rows, `/v1/query` and pipe results — are spelled by ClickHouse** (`date_time_output_format=iso`: RFC 3339 in UTC, `"2026-06-21T04:00:00.123Z"`, whatever zone the column declares, with the fraction at the column's own precision and trailing zeros kept: a `DateTime64(3)` on a whole second is `….000Z`, where 0.1.0's canonicalizer trimmed it to `…:00Z`; `Date` and `Date32` keep their `"2026-06-21"` form on the stream, and on `/v1/query` and pipes they lose the `T00:00:00Z` the native path added — see the structured-query entry) rather than canonicalized by a rewriting step in WaveHouse, so every surface agrees by construction (an event published before the upgrade, or by an older instance during a rolling deploy, replays from the stream in the spelling it was published in; closes [#372](https://github.com/Wave-RF/WaveHouse/issues/372) a different way than originally planned). The engine is one per process with a table set per tenant, bound from each tenant's own schema refresh: a tenant not bound yet, one whose ClickHouse line has no installed artifact, or one whose server time zone differs from the zone this process already opened that line with (one process serves one server time zone per ClickHouse line) is refused on its own, and so is a table whose schema chtypes could not compile (that table alone) — ingest answers `503` with `Retry-After: 5` and the generic body `{"error":"ingest validation is unavailable"}` (the cause, with zone names and artifact paths, goes to the server log only), and a stream whose role has a row `filter` withholds that tenant's (or that table's) rows with reason `unavailable` — while every other tenant keeps working. Each cause is checked again at every schema refresh and clears at the first one after it is fixed ([Deployment → chtypes artifacts](https://wavehouse.dev/deployment#chtypes-artifacts)). Row `filter` grants and insert `check` clauses are one mechanism now: both compile to a chtypes filter with every bound value a `{p:String}` parameter, and only a definite true admits — a compile failure, an evaluation error or a decline fails closed. Withheld stream rows are counted by `wavehouse_sse_rows_withheld_total{table,role,reason}` with `reason` one of `filter`, `error`, `decline`, `unavailable` and `drift`; a reader whose `filter` uses a column the inserting role cannot write (or a `MATERIALIZED` column) is declined every such row on the stream, though `/v1/query` returns them. The parse profile carries the type gates, so a table with `LowCardinality()`, a `FixedString` longer than 256 or a `Variant` column ingests and filters. A dedupe id is read from the exported row: a `null` cell or an empty string is missing (so is an omitted `String` id with no `DEFAULT`), and a numeric id column cannot tell an omitted `0` from a supplied one. A record whose insert grant resolved for another operation is a `403` for the whole request, an empty array (`[]`) included, where 0.1.0 answered `200`. Only `api`-role processes load the artifact: an API process refuses to start without one, an ingest-only or sweeper-only process needs none. - **A column the role may not insert is now ClickHouse's code 117, not a WaveHouse 403** (BREAKING; `internal/api/ingest.go`, `internal/typelayer/typelayer.go`, `clients/ts/src/types.ts`, `tests/e2e/sdk/ingest.test.ts`): column policy on the write path is answered by compiling the role its **own** copy of the table schema, instead of walking a decoded record's keys. A column the role may not write stays in that schema as `MATERIALIZED` of its default, so naming it is refused while expressions that read it keep working and the stored row holds the table's default. A record naming one is therefore refused by ClickHouse's parser exactly as an unknown column is — `400 {"exception_code":117,"error":"Unknown field found while parsing JSONEachRow format: x"}` (per record; a `header=present` header naming it fails the whole request with `code: "clickhouse.rejected"` and `exception_code: 117`) where 0.1.0 answered `403 {"error":"column \"x\" not allowed for insert"}`. The message no longer confirms whether the column exists, which is arguably the better answer. The read paths are unchanged: a denied column is still `403 column "x" not allowed` on `/v1/query` and still stripped from SSE events. Two further consequences of the same mechanism: an `_eq` insert check auto-injects by way of a `DEFAULT ''` on that compiled schema, so a supplied value still wins and an absent one is filled — but an `_in` check, which has no single value to stamp, now tests **the table's own default** against the claim-derived set rather than rejecting an absent column outright; and an explicit `null` on a non-`Nullable` checked column behaves exactly like omitting it (on a `Nullable` one it stores `NULL`, which fails the check). An `_eq` check also covers a column the role may not otherwise write: a record that omits it is filled with the required value and published, one that supplies exactly that value is accepted (**0.1.0 answered `403 column "x" not allowed for insert` to the correct value**), and any other value is `403 check failed for column "x"`. A role whose schema cannot be compiled this way, or that may write no column of the table, is refused with `500 {"error":"this role's insert permissions cannot be enforced on this table","retryable":false}` and no `Retry-After` (the cause is logged once a minute), rather than a `503` that a retry could not fix; a policy `check` on an `EPHEMERAL` column is still `403`. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 4efe625a..ec6af649 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -236,7 +236,7 @@ Validates a body of records against the ClickHouse schema for `{table}` and publ | `text/csv; header=present`, `text/tab-separated-values; header=present` | a header line naming the columns, in any order — see [Header formats](#header-formats-headerpresent) | | anything else, or none | `415`, listing the accepted types | -The two JSON families are one format to ClickHouse; the declaration decides only how the body frames its records. The single thing the body still chooses is *arity within `application/json`*: the first non-whitespace byte picks an array (`[`) or a single object. Under a single-object body only the first object is answered — concatenated objects after it are parsed but neither published nor reported, a `200` for one record (one cut off mid-record, or a short `UUID` in the first that takes the next with it, can still turn that answer into a `422` decline); declare NDJSON for anything line-framed ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). The reverse now works when every element is valid: a JSON array declared `application/x-ndjson` ingests every element. It is not re-framed, though, so one bad element in a single-line array makes chtypes decline the whole body (`422` for every record); declare `application/json` to get per-record answers. +The two JSON families are one format to ClickHouse; the declaration decides only how the body frames its records. The single thing the body still chooses is *arity within `application/json`*: the first non-whitespace byte picks an array (`[`) or a single object. It is found the way ClickHouse's JSON reader starts a body: past a UTF-8 byte order mark at the very start, with form feed and vertical tab counted as whitespace. Under a single-object body only the first object is answered — concatenated objects after it are parsed but neither published nor reported, a `200` for one record (one cut off mid-record, or a short `UUID` in the first that takes the next with it, can still turn that answer into a `422` decline); declare NDJSON for anything line-framed ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). The reverse now works when every element is valid: a JSON array declared `application/x-ndjson` ingests every element. It is not re-framed, though, so one bad element in a single-line array makes chtypes decline the whole body (`422` for every record); declare `application/json` to get per-record answers. :::note[What counts as a valid declaration] The header is parsed with Go's `mime.ParseMediaType` (RFC 9110 §8.3) and the **media type** decides the format, so no malformed *parameter* costs the request — `application/json; charset`, `application/json;;`, a value left mid-quote, a name repeated with different values all read as `application/json`. The one parameter that also decides a format is `header`, on `text/csv` and `text/tab-separated-values` only: `present` selects the header format, `absent` the strictly positional one, no `header` at all ClickHouse's default reading, and any other value is a `415`. A line whose parameters did not parse and that mentions `header` is a `415` as well, because guessing at it could ingest a declared header line as data or drop a data row as a header. Two more things are refused. A malformed parameter on a line that **also contains a comma** is a `415`, because the comma may be a second declaration joined on and the error cannot tell that from a comma inside data ([#563](https://github.com/Wave-RF/WaveHouse/issues/563)) — so `application/json; profile="a,b"` is fine and `application/json; profile="a,b"; charset` is not. And `Content-Type` is a **singleton** field (§5.3 forbids repeating it), so repeated header *lines* are accepted only when they agree, while a comma-joined value is refused outright: §8.3 warns that picking a member of the resulting pseudo-list is itself an interoperability and security hazard. @@ -323,7 +323,7 @@ WaveHouse rewrites timestamps in neither direction. **Inbound**, any spelling Cl | `text/csv; header=absent` | `CSV`, strictly positional: header detection is off (`input_format_csv_detect_header=0`), so every line is a record | | `text/csv` (no `header` parameter) | ClickHouse's default `CSV`: header auto-detection stays on | -`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. In either positional reading (no `header` parameter, or `header=absent`), the fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column, and, for a role with column restrictions, minus every column it may not write (an `_eq`-checked column stays) — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column, and only one that meets the [conditions above](#post-v1ingesttabletable--ingest-data). +`text/tab-separated-values` maps the same way, with `input_format_tsv_detect_header`. The auto-detection is ClickHouse's own heuristic, not WaveHouse's: send `header=absent` when a data row could spell the column names or you need the first line always read as a record. A leading UTF-8 byte order mark, which spreadsheet exports write, is ClickHouse's to read as well: it skips the mark ahead of a `header=present` header and ahead of most first columns (`UUID`, dates and times, numbers, `Enum`), but a `String` or `FixedString` first column (inside `Nullable` or `LowCardinality` too, or as the element of an `Array` or the key of a `Map`) keeps it as the start of the first value. So on a table whose first column is a string, a header line led by the mark is a record, not a detected header; send `header=present`, or drop the mark. A CSV line ends at LF, CRLF or LF CR, a TSV line at LF. In either positional reading (no `header` parameter, or `header=absent`), the fields are the table's **wire columns** — declaration order minus every `MATERIALIZED`, `ALIAS` and `EPHEMERAL` column, and, for a role with column restrictions, minus every column it may not write (an `_eq`-checked column stays) — and a producer must send **every one of them, in that order**. `GET /v1/ops/schema?table={table}` returns the columns in `position` order; drop the three kinds and that is the field order. Only `header=present` can name an `EPHEMERAL` column, and only one that meets the [conditions above](#post-v1ingesttabletable--ingest-data). | Body | Outcome | | --- | --- | diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index f69240f0..92737d20 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -85,7 +85,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. A read is cached and coalesced; a write — bound SQL that `IsMutation` (`sql_classify.go`) classifies as one — bypasses both and runs every call. `pipes.json` is the only way to define or change a pipe. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ch_errors.go** — `writeCHError`, the one mapping from a failed ClickHouse query to a response, shared by `/v1/query`, pipes and `/v1/ops/query` so they cannot drift apart: `chconn.Classify` decides the class, and the class the status, `code` and `retryable` ([ClickHouse errors on the query paths](/api#clickhouse-errors-on-the-query-paths)). A write pipe answers through `writeCHWriteError`, the same mapping with `retryable` always `false` and no `Retry-After`, since the write may have run. -- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (brackets that do not balance are a whole-request `400 invalid json: unterminated json array`, an empty element — a leading, doubled or trailing comma — a whole-request `400 invalid json: empty element in the json array …`, anything but whitespace after the closing `]` a whole-request `400 invalid json: content after the closing ']' of the json array`, and a `…WithNames` header naming an unknown or repeated column a whole-request `400 clickhouse.rejected` with code 117, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. Those verdicts account for every record or for none: the type layer counts the body's records itself (`typelayer/records.go`: the array's element count, passed in as `IngestOptions.Records`, or a floor over NDJSON objects and CSV/TSV lines less a header), because a value whose reader consumes past its own record — a `UUID` shorter than 36 characters — leaves the records after it with no verdict while the batch is still accepted, and when chtypes answers fewer (for an array, any other number) the whole body is declined (`Batch.Miscount`, one `422` per counted record, logged once) rather than reported short. A tenant or table the engine cannot answer for (the tenant not bound yet, no artifact for its ClickHouse line, a server time zone that differs from the one this process opened that line with, or a table schema the parser could not compile) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read, without affecting other tenants. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's duplicate window (two minutes on the embedded broker) — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). +- **ingest.go** — Accepts `POST /v1/ingest?table={table}` and hands the body to ClickHouse's own parser in one call. The **required** `Content-Type` chooses the format (`content_type.go`: the `application/json` and NDJSON spellings → `JSONEachRow`, `text/csv` → `CSV`, `text/tab-separated-values` → `TSV`, and each of those two with `; header=present` → `CSVWithNames` / `TSVWithNames`; `; header=absent` → the same formats with header detection off, a bare type leaves ClickHouse's auto-detection on, any other `header` value is a `415`); the bytes never choose it. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer, so the `413` lands before any record is processed. `ingest_framing.go` is the only code that reads those bytes itself: the first non-whitespace byte, found as ClickHouse's JSON reader finds it (past a leading UTF-8 byte order mark, with form feed and vertical tab as whitespace), answers the one remaining question inside the JSON family (array → batch response, otherwise single object), a top-level array is re-framed in place — outer brackets and depth-1 commas blanked to newlines — so one bad record cannot cost the batch (brackets that do not balance are a whole-request `400 invalid json: unterminated json array`, an empty element — a leading, doubled or trailing comma — a whole-request `400 invalid json: empty element in the json array …`, anything but whitespace after the closing `]` a whole-request `400 invalid json: content after the closing ']' of the json array`, and a `…WithNames` header naming an unknown or repeated column a whole-request `400 clickhouse.rejected` with code 117, nothing published), and the dedupe id is read positionally out of the exported row. One `IngestWith` call per body, on the role's own table (`Engine.RoleTable`, from the table set bound for the request's tenant) held only for the parse, parses, validates and checks in the same pass (the role's insert `check` clauses compiled into a row filter): it returns a verdict per record and the accepted rows as `JSONCompactEachRow` bytes, with no second parse for the checks. Those verdicts account for every record or for none: the type layer counts the body's records itself (`typelayer/records.go`: the array's element count, passed in as `IngestOptions.Records`, or a floor over NDJSON objects and CSV/TSV lines less a header, a leading byte order mark counted the way ClickHouse reads it for the table's first column), because a value whose reader consumes past its own record — a `UUID` shorter than 36 characters — leaves the records after it with no verdict while the batch is still accepted, and when chtypes answers fewer (for an array, any other number) the whole body is declined (`Batch.Miscount`, one `422` per counted record, logged once) rather than reported short. A tenant or table the engine cannot answer for (the tenant not bound yet, no artifact for its ClickHouse line, a server time zone that differs from the one this process opened that line with, or a table schema the parser could not compile) is a `503` with `Retry-After: 5` and the generic body `ingest validation is unavailable` (the cause goes to the log, rate-limited per tenant and table), decided before the body is read, without affecting other tenants. The accepted records run in windows of up to 256 (`ingestWindow`) through three phases: one dedupe `Reserve` for the window's ids, the publishes in record order (a deduped record under `mq.WithIdempotencyKey`, keyed by `dedupe.IdempotencyKey`), and one `Commit` of the published ids — a window is the unit of a dedupe round trip and of Pebble's commit `fsync`. An id another request holds answers `503` with the lease as `Retry-After`, a store that cannot answer (`dedupe.ErrUnavailable`) `503` with `Retry-After: 5`; a publish that fails at a record commits the ones before it and releases the rest, except that a failure other than `mq.ErrQueueFull` may have stored the event, so that record's claim is left to lapse and the idempotency key drops the retry's copy if it comes within the stream's duplicate window (two minutes on the embedded broker) — `mq.ErrUnavailable` (a broker blip) is one such failure, and still answers `503`: with the lease, rounded up to whole seconds, as `Retry-After` when the failing record held a claim left to lapse, else the flat `Retry-After: 5`. Each row goes through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row whose configured `id_field` cell is absent, `null` or an empty string can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` and `/` left unescaped via `output_format_json_escape_forward_slashes=0`, the same spellings the structured-query path and the SSE wire use, so a timestamp and a `/` read the same on every surface. - **clickhouse_http.go** — the reader behind `POST /v1/query` and `/v1/pipes/{name}`: it sends the statement to the resolved tenant's ClickHouse over HTTP (`chconn.Pools.Target`; the zero target — no pool — is a `503` with `Retry-After`) with `default_format=JSONEachRow` and every scalar filter value bound as a named `{pN:String}` parameter on the query string, so ClickHouse renders each row and WaveHouse only frames the lines into an array. A query with an `in` list goes as `multipart/form-data`: the SQL in the `query` field and each list as an external table (`_pN`, one `String` column in `RowBinary`), which ClickHouse's 128 KiB field limit does not touch; `checkRequestSize` answers `400` at the limits that remain. Every read carries fixed settings — `wait_end_of_query=1`, `http_write_exception_in_output_format=0`, a server-side `max_execution_time` (the smaller of the role's cap and the tenant's `query_timeout`), `cancel_http_readonly_queries_on_client_close=1`, and pinned rendering knobs (`output_format_json_quote_64bit_integers=0`, `output_format_json_quote_decimals=0`, `output_format_json_quote_denormals=0`, `output_format_json_named_tuples_as_objects=1`, `output_format_json_escape_forward_slashes=0`, `date_time_output_format=iso`, so a `/` is not escaped and a timestamp is RFC 3339 in UTC, both as on the SSE wire) — and `readonly=2` on reads (write pipes are the one exception). A failure comes back as a `chconn` HTTP error, so `chconn.Classify` and `writeCHError` apply as on every other ClickHouse path, and a response past the 64 MiB cap is `clickhouse.response_too_large`. The read paths hold at most the tenant's pool size in HTTP connections — its `max_open_conns`, or the largest among the tenants sharing its connection tuple (address, database, user, password, TLS) — one cap per tuple, like the native pool; tenants on one server with a different database, user, password or TLS settings each have their own. - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). diff --git a/internal/api/content_type.go b/internal/api/content_type.go index cb261a43..a88b3585 100644 --- a/internal/api/content_type.go +++ b/internal/api/content_type.go @@ -1,6 +1,7 @@ package api import ( + "bytes" "errors" "fmt" "mime" @@ -286,22 +287,38 @@ func mediaTypePrefix(v string) string { return base } -// firstNonSpace returns the body's first non-whitespace byte. ok is false when -// there is none within the sniff window — the same bound the streaming reader -// used, kept so a body of leading whitespace longer than the window still reads -// as empty rather than changing meaning now that the whole body is in memory. +// firstNonSpace returns the body's first non-whitespace byte, read the way +// ClickHouse's JSON reader reads the start of a body: a UTF-8 byte order mark +// is skipped at the very start (and only there), and whitespace is all six +// ASCII spaces, form feed and vertical tab included. Reading `\f[` or a +// BOM-led array as a single object published element 0 behind a 200 and +// dropped the rest. ok is false when there is no such byte within the sniff +// window — the same bound the streaming reader used, kept so a body of leading +// whitespace longer than the window still reads as empty rather than changing +// meaning now that the whole body is in memory. func firstNonSpace(body []byte) (byte, bool) { - for _, c := range body[:min(len(body), maxSniffBytes)] { - switch c { - case ' ', '\t', '\n', '\r': - continue - default: + window := body[:min(len(body), maxSniffBytes)] + for _, c := range bytes.TrimPrefix(window, utf8BOM) { + if !isJSONSpace(c) { return c, true } } return 0, false } +// utf8BOM is the UTF-8 byte order mark. +var utf8BOM = []byte("\xEF\xBB\xBF") + +// isJSONSpace reports whether ClickHouse's JSON reader skips c between values: +// the six ASCII whitespace bytes. +func isJSONSpace(c byte) bool { + switch c { + case ' ', '\t', '\n', '\r', '\f', '\v': + return true + } + return false +} + // emptyBodyMessage tailors the empty-body 400 message to the declared format so // an NDJSON caller still gets the familiar "empty ndjson body". func emptyBodyMessage(format IngestFormat) string { diff --git a/internal/api/ingest_count_test.go b/internal/api/ingest_count_test.go index b07e2cc2..2a6ed3ca 100644 --- a/internal/api/ingest_count_test.go +++ b/internal/api/ingest_count_test.go @@ -16,9 +16,17 @@ const testUUID = "61f0c404-5cb3-11e7-907b-a6006ad3dba0" // uuidRegistry holds the two tables a malformed UUID takes records from: // visits as a producer writes it (the id generated when omitted), and pings -// with the id first, for the positional formats. +// with the id first, for the positional formats. stamps is DateTime-first, a +// first column that skips a leading byte order mark like UUID does. func uuidRegistry(t *testing.T) *discovery.SchemaRegistry { return testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{ + { + Name: "stamps", + Columns: []discovery.Column{ + {Name: "ts", Type: "DateTime", Position: 1}, + {Name: "page", Type: "String", Position: 2}, + }, + }, { Name: "visits", Columns: []discovery.Column{ @@ -55,6 +63,10 @@ func TestIngest_ShortAnswerDeclinesTheWholeBatch(t *testing.T) { {"a header=absent CSV body", "pings", "text/csv; header=absent", "zzz,/a,1\n" + testUUID + ",/b,2\n" + testUUID + ",/c,3\n"}, {"a header=present CSV body", "pings", "text/csv; header=present", "id,page,n\nzzz,/a,1\n" + testUUID + ",/b,2\n" + testUUID + ",/c,3\n"}, {"a TSV body", "pings", "text/tab-separated-values", "zzz\t/a\t1\n" + testUUID + "\t/b\t2\n" + testUUID + "\t/c\t3\n"}, + {"a byte-order-marked CSV body", "pings", "text/csv", utf8BOMString + "zzz,/a,1\n" + testUUID + ",/b,2\n" + testUUID + ",/c,3\n"}, + // A String first column keeps the mark, so ClickHouse reads this + // names line as a record whose id is `id`, which takes /a with it. + {"a byte-order-marked header a String-first table reads as a record", "visits", "text/csv", utf8BOMString + "page,id\n/a," + testUUID + "\n/b," + testUUID + "\n"}, } { t.Run(tt.name, func(t *testing.T) { t.Parallel() @@ -125,6 +137,16 @@ func TestIngest_CountedBodiesKeepPerRecordVerdicts(t *testing.T) { {"a detected subset header", "pings", "text/csv", "page,id\n/a," + u + "\n", 1, 1}, {"a detected TSV header", "pings", "text/tab-separated-values", "id\tpage\tn\n" + u + "\t/a\t1\n", 1, 1}, {"CRLF line endings", "pings", "text/csv; header=absent", u + ",/a,1\r\n" + u + ",/b,2\r\n", 2, 2}, + {"LF CR line endings", "pings", "text/csv; header=absent", u + ",/a,1\n\r" + u + ",/b,2\n\r", 2, 2}, + {"LF CR line endings after a detected header", "pings", "text/csv", "id,page,n\n\r" + u + ",/a,1\n\r" + u + ",/b,2\n\r", 2, 2}, + // ClickHouse skips a leading byte order mark before a UUID or + // DateTime first column, so the names line after it is a header. + {"a byte order mark before a detected header", "pings", "text/csv", utf8BOMString + "id,page,n\n" + u + ",/a,1\n" + u + ",/b,2\n", 2, 2}, + {"a byte order mark before a DateTime-first header", "stamps", "text/csv", utf8BOMString + "ts,page\n2024-01-01 00:00:00,/a\n2024-01-02 00:00:00,/b\n", 2, 2}, + {"a byte order mark before a detected TSV header", "pings", "text/tab-separated-values", utf8BOMString + "id\tpage\tn\n" + u + "\t/a\t1\n" + u + "\t/b\t2\n", 2, 2}, + {"a byte order mark before a quoted newline", "visits", "text/csv; header=present", utf8BOMString + "\"page\",id\n\"/a\nx\"," + u + "\n/b," + u + "\n", 2, 2}, + {"a byte order mark before a header=absent body", "pings", "text/csv; header=absent", utf8BOMString + u + ",/a,1\n" + u + ",/b,2\n", 2, 2}, + {"a byte order mark before an NDJSON body", "pings", "application/x-ndjson", utf8BOMString + "{\"page\":\"/a\"}\n{\"page\":\"/b\"}\n", 2, 2}, } { t.Run(tt.name, func(t *testing.T) { t.Parallel() @@ -145,6 +167,69 @@ func TestIngest_CountedBodiesKeepPerRecordVerdicts(t *testing.T) { } } +// utf8BOMString is a UTF-8 byte order mark, as an editor or a spreadsheet +// export leads a file with it. +const utf8BOMString = "\xEF\xBB\xBF" + +// TestIngest_JSONArrayLedByAMarkOrAFormFeed is the regression guard for a +// silent loss: the array-or-object byte skipped only space, tab, CR and LF, so +// an array led by a byte order mark or a form feed took the single-object path +// and published element 0 behind `200 {"ok":true}`. ClickHouse's JSON reader +// skips the mark at the start and all six ASCII whitespace bytes, so these are +// arrays: every element is answered and published. +func TestIngest_JSONArrayLedByAMarkOrAFormFeed(t *testing.T) { + t.Parallel() + for _, tt := range []struct{ name, body string }{ + {"a byte order mark", utf8BOMString + `[{"page":"/a"},{"page":"/b"},{"page":"/c"}]`}, + {"a form feed", "\f" + `[{"page":"/a"},{"page":"/b"},{"page":"/c"}]`}, + {"a vertical tab", "\v" + `[{"page":"/a"},{"page":"/b"},{"page":"/c"}]`}, + {"a pretty-printed, marked array", utf8BOMString + "\n[\n {\"page\": \"/a\"},\n\f {\"page\": \"/b\"},\n {\"page\": \"/c\"}\v\n]\f\n"}, + } { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, uuidRegistry(t), pub) + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "pings", "application/json", tt.body))) + + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + resp := decodeBatchResult(t, w) + assert.Equal(t, 3, resp.Total, "the batch response, not the single-object one") + assert.Equal(t, 3, resp.Succeeded) + require.Len(t, pub.Messages, 3, "every element is published") + for i, page := range []string{"/a", "/b", "/c"} { + assert.Equal(t, page, publishedRow(t, pub.Messages[i].Data)["page"]) + } + }) + } +} + +// TestIngest_MarkedSingleObjectAndEmptyBodies: a single object led by a mark is +// still one object, and a body that is only a mark and whitespace is empty. +func TestIngest_MarkedSingleObjectAndEmptyBodies(t *testing.T) { + t.Parallel() + pub := &testutil.MockPublisher{} + h := newTestIngestHandler(t, uuidRegistry(t), pub) + + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "pings", "application/json", utf8BOMString+"\f"+`{"page":"/a"}`))) + require.Equal(t, http.StatusOK, w.Code, "body=%s", w.Body.String()) + assert.JSONEq(t, `{"ok":true}`, w.Body.String()) + require.Len(t, pub.Messages, 1) + + for _, tt := range []struct{ contentType, message string }{ + {"application/json", "empty body"}, + {"application/x-ndjson", "empty ndjson body"}, + {"text/csv", "empty csv body"}, + } { + w := httptest.NewRecorder() + h.Handle(w, withTenant(rawIngestRequest(t, "pings", tt.contentType, utf8BOMString+" \f\v\n"))) + require.Equal(t, http.StatusBadRequest, w.Code, "%s: body=%s", tt.contentType, w.Body.String()) + assert.Equal(t, tt.message, jsonErrorMessage(t, w), tt.contentType) + } + assert.Len(t, pub.Messages, 1, "nothing more is published") +} + // TestIngest_EmptyArrayElementIsInvalidJSON: a leading, doubled or trailing // comma frames as a blank line, which is no record to ClickHouse, so the // element count could not hold. It is refused as the invalid JSON it is. diff --git a/internal/api/ingest_framing.go b/internal/api/ingest_framing.go index ca0c2340..65b61ecd 100644 --- a/internal/api/ingest_framing.go +++ b/internal/api/ingest_framing.go @@ -1,6 +1,7 @@ package api import ( + "bytes" "encoding/json" "errors" ) @@ -23,8 +24,10 @@ import ( // // The scan is string- and escape-aware, so a comma or a bracket inside a value // is untouched, and it runs ONLY when the declared format is the JSON family and -// the first non-whitespace byte is '['. It must not run on anything else: a bare -// object's own commas are at depth 1 and rewriting them destroys the record +// the first non-whitespace byte is '[' (see firstNonSpace: a leading byte order +// mark and form feeds count as whitespace there, so they do here too, and stay +// in place for ClickHouse's reader to skip). It must not run on anything else: a +// bare object's own commas are at depth 1 and rewriting them destroys the record // (measured). // // Three substitutions, all in place and all the same length: @@ -60,11 +63,15 @@ func reframeArray(b []byte) (elements int, err error) { sawValue, closed := false, false element := false // a value since the opening bracket or the last depth-1 comma inStr, esc := false, false - for i := range b { + start := 0 + if bytes.HasPrefix(b, utf8BOM) { + start = len(utf8BOM) + } + for i := start; i < len(b); i++ { c := b[i] if closed { switch c { - case ' ', '\t': + case ' ', '\t', '\f', '\v': case '\n', '\r': b[i] = ' ' default: @@ -110,7 +117,7 @@ func reframeArray(b []byte) (elements int, err error) { element = false case c == '\n' || c == '\r': b[i] = ' ' - case c == ' ' || c == '\t': + case isJSONSpace(c): default: sawValue = sawValue || depth >= 1 element = element || depth >= 1 diff --git a/internal/api/ingest_framing_test.go b/internal/api/ingest_framing_test.go index 226f3758..4c887ee6 100644 --- a/internal/api/ingest_framing_test.go +++ b/internal/api/ingest_framing_test.go @@ -1,6 +1,7 @@ package api import ( + "strings" "testing" "github.com/stretchr/testify/assert" @@ -92,6 +93,21 @@ func TestReframeArray(t *testing.T) { want: " {\"a\":1} \t", count: 1, ok: true, }, + { + name: "a leading byte order mark stays for ClickHouse to skip", + body: "\xEF\xBB\xBF[{\"a\":1},{\"a\":2}]", + want: "\xEF\xBB\xBF {\"a\":1}\n{\"a\":2} ", + count: 2, ok: true, + }, + { + name: "form feeds and vertical tabs are layout", + body: "\f[\v{\"a\":1},\f{\"a\":2}\v]\f\v", + want: "\f \v{\"a\":1}\n\f{\"a\":2}\v \f\v", + count: 2, ok: true, + }, + {name: "an array of layout alone is empty", body: "[\f\v ]", want: " \f\v ", count: 0, ok: true}, + {name: "a form feed is no element", body: "[{\"a\":1},\f]", err: errEmptyElement}, + {name: "a byte order mark after the array is content", body: "[{\"a\":1}]\xEF\xBB\xBF", err: errAfterArray}, {name: "a truncated array does not balance", body: `[{"a":1}`, err: errUnterminatedArray}, {name: "a trailing comma cut off does not balance", body: `[{"a":1},`, err: errUnterminatedArray}, {name: "a bare open bracket does not balance", body: `[`, err: errUnterminatedArray}, @@ -141,6 +157,30 @@ func TestReframeArray_DestroysWhatItMustNotSee(t *testing.T) { "an NDJSON body is destroyed too — same reason, same gate") } +// TestFirstNonSpace: the array-or-object byte is found the way ClickHouse's +// JSON reader starts a body — a byte order mark skipped at the very start only, +// and all six ASCII whitespace bytes skipped. +func TestFirstNonSpace(t *testing.T) { + t.Parallel() + for body, want := range map[string]byte{ + "[": '[', + " \t\r\n\f\v[": '[', + "\xEF\xBB\xBF[": '[', + "\xEF\xBB\xBF\f\n {": '{', + " \xEF\xBB\xBF[": 0xEF, // a mark after layout is not skipped, by ClickHouse either + "\xEF\xBB\xBF\xEF\xBB\xBF[": 0xEF, + "\xEF\xBBx": 0xEF, + } { + got, ok := firstNonSpace([]byte(body)) + assert.True(t, ok, "%q", body) + assert.Equal(t, want, got, "%q", body) + } + for _, body := range []string{"", " \f\v\r\n\t", "\xEF\xBB\xBF", "\xEF\xBB\xBF\n", strings.Repeat(" ", maxSniffBytes) + "["} { + _, ok := firstNonSpace([]byte(body)) + assert.False(t, ok, "%q reads as empty", body) + } +} + func TestCellAt(t *testing.T) { t.Parallel() const line = `["a, b", "c\"d", 42, null, ["x", "y"], {"k": 1}, "last"]` diff --git a/internal/typelayer/ingest.go b/internal/typelayer/ingest.go index a2fcffae..03fbdad4 100644 --- a/internal/typelayer/ingest.go +++ b/internal/typelayer/ingest.go @@ -343,13 +343,15 @@ func span(res chtypes.BatchResult, i int) []byte { // floor is recordFloor for body against this handle's columns: header // auto-detection compares a first line with the wire columns and a second with -// their compiled types. Skipped when the caller has an exact count. +// their compiled types, and the first column's type decides whether a leading +// byte order mark is read as framing. Skipped when the caller has an exact +// count. func (t *Table) floor(s *schemaSlot, format Format, opts IngestOptions, body []byte) int { if opts.Records > 0 { return 0 } var types map[string]string - if (format == FormatCSV || format == FormatTSV) && !opts.StrictPositional { + if format == FormatCSV || format == FormatTSV { types = make(map[string]string, len(s.schema.Columns)) for _, c := range s.schema.Columns { types[c.Name] = c.Type diff --git a/internal/typelayer/records.go b/internal/typelayer/records.go index 21f47640..6f8bdf62 100644 --- a/internal/typelayer/records.go +++ b/internal/typelayer/records.go @@ -3,6 +3,7 @@ package typelayer import ( "bytes" "strconv" + "strings" ) // Counting the records a body holds, independently of chtypes. @@ -33,9 +34,11 @@ import ( // wider one; measured), as is a non-empty tail with no newline. A CSV // newline inside a double-quoted field, a quote opening after spaces or // tabs included, is not a terminator; a quote anywhere else is a literal. -// A TSV newline escaped by a backslash is not one either. The header line -// of a WithNames body is not a record; under header auto-detection see -// detectedHeaderRows. +// A TSV newline escaped by a backslash is not one either. A CSV line ends +// at LF, CRLF or LF CR; a TSV one at LF alone (an LF CR leaves the CR to +// open the next record). The header line of a WithNames body is not a +// record; under header auto-detection see detectedHeaderRows. For a +// leading UTF-8 byte order mark see bomFloor. func recordFloor(format Format, opts IngestOptions, body []byte, wire []string, types map[string]string) int { switch format { case FormatJSONEachRow: @@ -44,6 +47,58 @@ func recordFloor(format Format, opts IngestOptions, body []byte, wire []string, default: return 0 } + if bytes.HasPrefix(body, utf8BOM) { + return bomFloor(format, opts, body, wire, types) + } + return lineFloor(format, opts, body, wire, types) +} + +// utf8BOM is the UTF-8 byte order mark. +var utf8BOM = []byte("\xEF\xBB\xBF") + +// bomFloor is the floor of a CSV or TSV body led by a UTF-8 byte order mark. +// Measured on 26.8.15.10, ClickHouse always skips the mark in a WithNames +// body; in a positional one (bare or header=absent) the first wire column's +// type decides. String and FixedString, bare or inside Nullable or +// LowCardinality, keep it as the start of the value, as do Array(String) and +// Map(String, …); UUID, DateTime, numbers, Enum, IPv4 and Array(UInt8) skip +// it. So +// `BOM id,page,n` is a detected header on a UUID- or DateTime-first table and +// a record on a String-first one, and a quote right after the mark opens a +// field only where it was skipped. +// +// A WithNames body is counted without the mark and one whose first column is +// a stringType with it. Every other first column takes the lower of the two readings, so the +// floor holds whichever one ClickHouse makes for a type not measured here. +func bomFloor(format Format, opts IngestOptions, body []byte, wire []string, types map[string]string) int { + skipped := lineFloor(format, opts, body[len(utf8BOM):], wire, types) + if format == FormatCSVWithNames || format == FormatTSVWithNames { + return skipped + } + kept := lineFloor(format, opts, body, wire, types) + if len(wire) > 0 && stringType(types[wire[0]]) { + return kept + } + return min(skipped, kept) +} + +// stringType reports whether t is String or FixedString, bare or inside +// Nullable or LowCardinality. +func stringType(t string) bool { + for { + if inner, ok := strings.CutPrefix(t, "Nullable("); ok { + t = strings.TrimSuffix(inner, ")") + } else if inner, ok := strings.CutPrefix(t, "LowCardinality("); ok { + t = strings.TrimSuffix(inner, ")") + } else { + return t == "String" || strings.HasPrefix(t, "FixedString(") + } + } +} + +// lineFloor is recordFloor for a CSV or TSV body: its records less the header +// lines ClickHouse reads as one. +func lineFloor(format Format, opts IngestOptions, body []byte, wire []string, types map[string]string) int { csv := format == FormatCSV || format == FormatCSVWithNames var n int var first, second []byte @@ -102,9 +157,10 @@ func jsonObjects(b []byte) int { // csvRecords counts ClickHouse's CSV records in b and returns the first two, // without their terminators. A double quote opens a quoted field only where a // field starts (leading spaces and tabs aside); inside one, `""` is a quote and -// a newline is data. ClickHouse reads the same body the same way (measured, -// including a quote after leading whitespace, a quote mid-field and a CRLF -// inside quotes). +// a newline is data. A CR right after a terminating LF belongs to it: LF CR is +// one line end to ClickHouse, like CRLF. ClickHouse reads the same body the +// same way (measured, including a quote after leading whitespace, a quote +// mid-field, a CRLF inside quotes and LF CR line ends). func csvRecords(b []byte) (n int, first, second []byte) { start := 0 inQuote, fieldStart := false, true @@ -129,6 +185,9 @@ func csvRecords(b []byte) (n int, first, second []byte) { case ' ', '\t': case '\n': n, first, second = takeRecord(n, first, second, b[start:i]) + if i+1 < len(b) && b[i+1] == '\r' { + i++ + } start, fieldStart = i+1, true default: fieldStart = false @@ -140,9 +199,10 @@ func csvRecords(b []byte) (n int, first, second []byte) { return n, first, second } -// tsvRecords is csvRecords for TSV: no quoting, and a newline after an odd run -// of backslashes is an escaped newline inside a field (measured: `a\` + LF is -// one value, `a\\` + LF ends the record). +// tsvRecords is csvRecords for TSV: no quoting, a newline after an odd run of +// backslashes is an escaped newline inside a field (measured: `a\` + LF is one +// value, `a\\` + LF ends the record), and a CR after an LF is not part of the +// line end — it is the first byte of the next record (measured). func tsvRecords(b []byte) (n int, first, second []byte) { start, run := 0, 0 for i, c := range b { diff --git a/internal/typelayer/records_test.go b/internal/typelayer/records_test.go index e0b33a99..17dad556 100644 --- a/internal/typelayer/records_test.go +++ b/internal/typelayer/records_test.go @@ -13,12 +13,24 @@ import ( "github.com/Wave-RF/WaveHouse/internal/tenant" ) -const recUUID = "61f0c404-5cb3-11e7-907b-a6006ad3dba0" +const ( + recUUID = "61f0c404-5cb3-11e7-907b-a6006ad3dba0" + bom = "\xEF\xBB\xBF" +) -// uuidTables are a UUID-first table for the positional formats and a -// name-addressed one whose omitted id takes a constant DEFAULT. +// uuidTables are a UUID-first table for the positional formats, a +// name-addressed one whose omitted id takes a constant DEFAULT, and two more +// first-column types a leading byte order mark is measured against. func uuidTables() []*discovery.TableSchema { return []*discovery.TableSchema{ + {Name: "stamps", Columns: []discovery.Column{ + {Name: "ts", Type: "DateTime", Position: 1}, + {Name: "page", Type: "String", Position: 2}, + }}, + {Name: "batches", Columns: []discovery.Column{ + {Name: "ns", Type: "Array(UInt8)", Position: 1}, + {Name: "page", Type: "String", Position: 2}, + }}, {Name: "pings", Columns: []discovery.Column{ {Name: "id", Type: "UUID", Position: 1}, {Name: "page", Type: "String", Position: 2}, @@ -67,6 +79,11 @@ func TestIngest_ShortAnswerIsDeclinedWhole(t *testing.T) { {"a blank CSV line", pings, FormatCSV, IngestOptions{StrictPositional: true}, recUUID + ",/a,1\n\n" + recUUID + ",/b,2\n", 3}, {"a CSVWithNames types line", pings, FormatCSVWithNames, IngestOptions{}, "id,page,n\nUUID,String,UInt8\n" + recUUID + ",/b,2\n", 2}, {"TSV", pings, FormatTSV, IngestOptions{}, "zzz\t/a\t1\n" + recUUID + "\t/b\t2\n" + recUUID + "\t/c\t3\n", 3}, + {"a byte order mark before bare CSV", pings, FormatCSV, IngestOptions{}, bom + "zzz,/a,1\n" + recUUID + ",/b,2\n" + recUUID + ",/c,3\n", 3}, + // A String first column keeps the mark as its value, so this names + // line is no header to ClickHouse: it is a record whose id is `id`. + {"a marked header a String-first table reads as a record", visits, FormatCSV, IngestOptions{}, bom + "page,id\n/a," + recUUID + "\n/b," + recUUID + "\n", 3}, + {"a marked TSV header a String-first table reads as a record", visits, FormatTSV, IngestOptions{}, bom + "page\tid\n/a\t" + recUUID + "\n/b\t" + recUUID + "\n", 3}, } { t.Run(tt.name, func(t *testing.T) { batch, err := tt.tbl.IngestWith(tt.format, tt.opts, []byte(tt.body)) @@ -97,8 +114,9 @@ func TestIngest_WholeAnswersAreNotMiscounted(t *testing.T) { t.Cleanup(tbl.Release) return tbl } - visits, pings := table("visits"), table("pings") + visits, pings, stamps, batches := table("visits"), table("pings"), table("stamps"), table("batches") r := recUUID + ",/r,1\n" + ts := "2024-01-01 00:00:00" for _, tt := range []struct { name string tbl *Table @@ -126,6 +144,21 @@ func TestIngest_WholeAnswersAreNotMiscounted(t *testing.T) { {"a detected TSV header with an escape", pings, FormatTSV, IngestOptions{}, "i\\x64\tpage\tn\n" + recUUID + "\t/r\t1\n", 1}, {"a CSVWithNames header", pings, FormatCSVWithNames, IngestOptions{}, "page,id\n/a," + recUUID + "\n/b," + recUUID + "\n", 2}, {"a header with no records", pings, FormatCSV, IngestOptions{}, "id,page,n\n", 0}, + // ClickHouse skips a leading byte order mark unless the first column + // is a string type, so the names line after it is a detected header. + {"a byte order mark before a detected header", pings, FormatCSV, IngestOptions{}, bom + "id,page,n\n" + r + r, 2}, + {"a byte order mark before a DateTime-first header", stamps, FormatCSV, IngestOptions{}, bom + "ts,page\n" + ts + ",/a\n" + ts + ",/b\n", 2}, + {"a byte order mark before a detected TSV header", pings, FormatTSV, IngestOptions{}, bom + "id\tpage\tn\n" + recUUID + "\t/a\t1\n" + recUUID + "\t/b\t2\n", 2}, + {"a byte order mark before a CSVWithNames header", visits, FormatCSVWithNames, IngestOptions{}, bom + "\"page\",id\n\"/a\nx\"," + recUUID + "\n", 1}, + {"a byte order mark before a quoted newline", batches, FormatCSV, IngestOptions{StrictPositional: true}, bom + "\"[1,\n2]\",/a\n[3],/b\n", 2}, + // A String first column keeps it: the quote after it is a literal, so + // its newline ends a record — three records, as ClickHouse reads them. + {"a byte order mark a String first column keeps", visits, FormatCSV, IngestOptions{StrictPositional: true}, bom + "\"/a\nx\"," + recUUID + "\n/b," + recUUID + "\n", 3}, + {"LF CR line ends", pings, FormatCSV, IngestOptions{StrictPositional: true}, recUUID + ",/a,1\n\r" + recUUID + ",/b,2\n\r", 2}, + {"LF CR after a detected header", visits, FormatCSV, IngestOptions{}, "page,id\n\r/a," + recUUID + "\n\r/b," + recUUID + "\n\r", 2}, + {"LF CR after a CSVWithNames header", visits, FormatCSVWithNames, IngestOptions{}, "page,id\n\r/a," + recUUID + "\n\r/b," + recUUID + "\n\r", 2}, + // TSV has no LF CR line end: the CR opens the next record. + {"a TSV CR after LF", visits, FormatTSV, IngestOptions{StrictPositional: true}, "/a\t" + recUUID + "\n\r/b\t" + recUUID + "\n", 2}, } { t.Run(tt.name, func(t *testing.T) { batch, err := tt.tbl.IngestWith(tt.format, tt.opts, []byte(tt.body)) @@ -178,6 +211,10 @@ func TestCSVRecords(t *testing.T) { {"a\"b\nc\",1\n", 2, "a\"b", "c\",1"}, {"'a\nb',1\n", 2, "'a", "b',1"}, {"a\r\nb\r\n", 2, "a\r", "b\r"}, + {"a\n\rb\n\r", 2, "a", "b"}, + {"a\n\r\nb\n", 3, "a", ""}, + {"a\n\r\rb\n", 2, "a", "\rb"}, + {"\"x\n\ry\"\n\rb", 2, "\"x\n\ry\"", "b"}, {"\"never closed\nstill quoted", 1, "\"never closed\nstill quoted", ""}, } { n, first, second := csvRecords([]byte(tt.body)) @@ -197,6 +234,7 @@ func TestTSVRecords(t *testing.T) { "a\\\\\\\nb\nc\n": 2, "a\n\nb": 3, "\"a\nb\"\n": 2, + "a\n\rb\n\r": 3, } { n, _, _ := tsvRecords([]byte(body)) assert.Equal(t, want, n, "%q", body) @@ -254,6 +292,48 @@ func TestRecordFloor(t *testing.T) { assert.Equal(t, 0, recordFloor(chtypes.JSONCompactEachRow, IngestOptions{}, []byte("[1]\n"), nil, nil), "a format Ingest does not parse") } +// TestRecordFloor_ByteOrderMark: a leading mark is counted the way ClickHouse +// reads it — skipped under a header, kept as a String first column's value — +// and as the lower of the two readings before any other first column. +func TestRecordFloor_ByteOrderMark(t *testing.T) { + t.Parallel() + wire := []string{"id", "page", "n"} + typed := func(first string) map[string]string { + return map[string]string{"id": first, "page": "String", "n": "UInt8"} + } + header := []byte(bom + "id,page,n\na\nb\n") + for _, tt := range []struct { + first string + want int + }{ + {"UUID", 2}, + {"Nullable(UUID)", 2}, + {"String", 3}, + {"FixedString(36)", 3}, + {"Nullable(String)", 3}, + {"LowCardinality(String)", 3}, + {"LowCardinality(Nullable(String))", 3}, + {"Array(String)", 2}, // keeps the mark too; the lower reading + {"", 2}, + } { + assert.Equal(t, tt.want, recordFloor(FormatCSV, IngestOptions{}, header, wire, typed(tt.first)), tt.first) + assert.Equal(t, tt.want, recordFloor(FormatTSV, IngestOptions{}, []byte(strings.ReplaceAll(string(header), ",", "\t")), wire, typed(tt.first)), "TSV "+tt.first) + } + assert.Equal(t, 2, recordFloor(FormatCSVWithNames, IngestOptions{}, header, wire, typed("String")), "a header always skips it") + + // A quote right after the mark opens a field only where the mark is + // skipped; the lower reading holds where that is not known. + quoted := []byte(bom + "\"a\nb\",x\nc\n") + assert.Equal(t, 2, recordFloor(FormatCSV, IngestOptions{StrictPositional: true}, quoted, wire, typed("UUID"))) + assert.Equal(t, 3, recordFloor(FormatCSV, IngestOptions{StrictPositional: true}, quoted, wire, typed("String"))) + // Here the skipped reading is the higher one (`"""` opens a field only + // without the mark), and a non-string first column takes the lower. + raised := []byte(bom + "\"\"\",\"\n\"") + assert.Equal(t, 1, recordFloor(FormatCSV, IngestOptions{StrictPositional: true}, raised, wire, typed("UUID"))) + assert.Equal(t, 1, recordFloor(FormatCSV, IngestOptions{StrictPositional: true}, raised, wire, typed("String"))) + assert.Equal(t, 0, recordFloor(FormatCSV, IngestOptions{StrictPositional: true}, []byte(bom), wire, typed("UUID"))) +} + func TestMiscount(t *testing.T) { t.Parallel() assert.Nil(t, miscount(0, 3, 3))