From 295d329718b365c68bd981fa939a7781810fd871 Mon Sep 17 00:00:00 2001 From: soustruh Date: Thu, 24 Sep 2026 12:46:50 +0200 Subject: [PATCH] chore(release): 0.95.0 --- .claude-plugin/marketplace.json | 2 +- CLAUDE.md | 2 +- docs/auth.md | 2 +- plugins/kbagent/.claude-plugin/plugin.json | 2 +- plugins/kbagent/agents/keboola-expert.md | 6 ++-- .../kbagent/references/auth-workflow.md | 4 +-- .../kbagent/references/commands-reference.md | 10 +++--- .../skills/kbagent/references/gotchas.md | 35 ++++++++++++------ .../references/semantic-layer-workflow.md | 4 +-- pyproject.toml | 2 +- src/keboola_agent_cli/changelog.py | 36 +++++++++++++++++++ src/keboola_agent_cli/commands/context.py | 6 ++-- uv.lock | 2 +- 13 files changed, 82 insertions(+), 31 deletions(-) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 5d7823c70..914eabdda 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -10,7 +10,7 @@ "plugins": [ { "name": "kbagent", - "version": "0.94.0", + "version": "0.95.0", "source": "./plugins/kbagent", "description": "DEPRECATED — install from keboola/ai-kit: /plugin marketplace add keboola/ai-kit && /plugin install kbagent@keboola-claude-kit — AI-friendly interface to Keboola Connection projects — explore configs, jobs, lineage, sync configs as files, manage dev branches, and debug SQL in workspaces", "category": "development" diff --git a/CLAUDE.md b/CLAUDE.md index 44a0d0061..0b8b09745 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -450,7 +450,7 @@ kbagent auth register-projects [--stack URL|alias] [--all] [--project-id ID ...] # See docs/web-server.md. kbagent project create --url URL [--project ALIAS] [--name NAME] [--backend snowflake|bigquery] [--sync-backend-init] -# project create (since vNEXT, DMD-1940): the ONLY kbagent command that works from nothing -- +# project create (since 0.95.0, DMD-1940): the ONLY kbagent command that works from nothing -- # no account, no token, no `auth login`. POSTs the unauthenticated provisioning endpoint # (`/manage/programmatic-projects`, gated by the `agent-provisioning` stack feature # / `STACK_FEATURES__AGENT_PROVISIONING`, off on most stacks -- the COMMAND is always diff --git a/docs/auth.md b/docs/auth.md index 1c80aac19..87e4ef92f 100644 --- a/docs/auth.md +++ b/docs/auth.md @@ -97,7 +97,7 @@ The reverse direction *is* possible, because it is an explicit request — see ### `project create` -- starting from nothing -*(since vNEXT)* +*(since 0.95.0)* ```bash kbagent project create --url URL [--project ALIAS] [--name NAME] \ diff --git a/plugins/kbagent/.claude-plugin/plugin.json b/plugins/kbagent/.claude-plugin/plugin.json index 21a0994e0..822197443 100644 --- a/plugins/kbagent/.claude-plugin/plugin.json +++ b/plugins/kbagent/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "kbagent", - "version": "0.94.0", + "version": "0.95.0", "description": "AI-friendly interface to Keboola Connection projects — explore configs, jobs, lineage, sync configs as files, manage dev branches, and debug SQL in workspaces", "author": { "name": "Keboola", diff --git a/plugins/kbagent/agents/keboola-expert.md b/plugins/kbagent/agents/keboola-expert.md index e88f8aad7..6a00d8f12 100644 --- a/plugins/kbagent/agents/keboola-expert.md +++ b/plugins/kbagent/agents/keboola-expert.md @@ -154,7 +154,7 @@ been retired, so its absence is NOT a promise (see §1 Rule 6). | Read a semantic-layer model (models, metrics, datasets, constraints) | `kbagent --json semantic-layer show --project P [--model M] [--type metric\|dataset\|relationship\|constraint\|glossary]`; `model list`; `search-context` / `get-context` for glob/id lookup; `validate [--deep]` before trusting one. The WHOLE `semantic-layer` family needs a MASTER token: a valid non-master token gets `MISSING_MASTER_TOKEN` (0.92.0+, #711; the Metastore's opaque 401 "Failed to create project scope") -- register a master token, do not escalate | -- | hand-rolled `httpx` loops against `metastore.*.keboola.com` (bypasses retry/backoff and the kbagent error envelope) | | ANY semantic-layer write (add / edit / remove / import / promote / build) | `kbagent semantic-layer export` FIRST (the metastore has no soft-delete and no version history -- the snapshot is the only restore path), then the write, `--dry-run` where offered. `Read` [semantic-layer-workflow.md](../skills/kbagent/references/semantic-layer-workflow.md) before starting: it carries the per-verb recipes, the rename cascade and the promote classification | `semantic-layer diff` (`--project-a/-b` or `--file-a/-b`) to confirm what a write would change | raw metastore REST (no rollback, no orphan scan, no modelUUID rewrite); a write with no export taken | | User asks to "log in" / authenticate via browser / register a session's projects | ATTENDED session + a background shell: run `kbagent auth login --device-code --stack URL --register-projects` in a **BACKGROUND** shell (capture stdout+stderr, human mode -- `--json` puts the panel on stderr), relay the verification URL + user code to the user, then poll `kbagent --json auth status` (exit 0 = signed in, exit 3 = not yet). The human's part is approving in the browser, not typing the command. No background shell -> hand the plain `auth login --stack URL --register-projects` to the user's terminal. To register projects from an EXISTING session, `kbagent auth register-projects --all` or `--project-id ID` is non-interactive and agent-safe | -- | running `auth login` in a FOREGROUND tool shell (~120 s timeout kills it mid-flight); running it unattended (nobody can approve); re-running it blind without checking `auth status` first (orphans a session); the flagless `register-projects` picker unattended; reading the token out of `auth.json`; using a numeric project id as an alias | -| User has NO Keboola account / project at all and asks to get started | `kbagent project create --url URL [--project ALIAS] [--name NAME] [--backend snowflake\|bigquery]` (vNEXT+) -- provisions a real project, stores its session and registers the alias in one call, agent-runnable. **Then relay the result's `confirm_url` to the human verbatim and say the project is owned by nobody until they open it**; after they confirm, the agent session is revoked by design -> `kbagent auth login --stack URL` | the user creating the project in the Keboola UI, then `auth login` / `project add --token` | calling it on a stack where a session already exists (exit 5 -- one session per stack); reporting success without the confirm link (an unclaimed project is a billable orphan); retrying it after a 5xx/429/503 without being asked (each success creates a new organization + project + credit grant) | +| User has NO Keboola account / project at all and asks to get started | `kbagent project create --url URL [--project ALIAS] [--name NAME] [--backend snowflake\|bigquery]` (0.95.0+) -- provisions a real project, stores its session and registers the alias in one call, agent-runnable. **Then relay the result's `confirm_url` to the human verbatim and say the project is owned by nobody until they open it**; after they confirm, the agent session is revoked by design -> `kbagent auth login --stack URL` | the user creating the project in the Keboola UI, then `auth login` / `project add --token` | calling it on a stack where a session already exists (exit 5 -- one session per stack); reporting success without the confirm link (an unclaimed project is a billable orphan); retrying it after a 5xx/429/503 without being asked (each success creates a new organization + project + credit grant) | | CI task has account credentials | `kbagent auth login-password --email E (--password-stdin \| --password P) [--totp-secret SEED]` (0.84.0+), agent-runnable | a static Storage token | `auth login` unattended | If the table does not cover the user's task, **ask clarifying @@ -395,7 +395,7 @@ its absence is NOT a promise the entry is version-independent (see §1 Rule 6). hand the plain command to the user, then `auth status`/`auth logout` as usual. **`auth login-password` (0.84.0+) IS the headless path** -- email + password (+ TOTP seed), agent-runnable; WebAuthn-only -> `AUTH_MFA_INVALID`. -- **`project create` (vNEXT+) is the only command that works from nothing** +- **`project create` (0.95.0+) is the only command that works from nothing** -- no account, no token, no `auth login` first; it needs the `agent-provisioning` stack feature (`STACK_FEATURES__AGENT_PROVISIONING`), which is OFF on most stacks. The COMMAND is always registered, so its @@ -450,7 +450,7 @@ its absence is NOT a promise the entry is version-independent (see §1 Rule 6). (`fallback_used: "heuristic"`), not the full AI wizard (that is the `sl-build` skill). - dataset `fqn` = the table's Storage location (`storage table-detail` -> - `sql_path`), since vNEXT: `add dataset` fails on a table that does not exist + `sql_path`), since 0.95.0: `add dataset` fails on a table that does not exist unless `--fqn` is given. Older kbagent wrote a `"KEBOOLA"` database that resolves nowhere -- `validate --deep` flags those as `FQN_MISMATCH`; never hand-build an fqn from the tableId (a linked bucket lives in the SOURCE diff --git a/plugins/kbagent/skills/kbagent/references/auth-workflow.md b/plugins/kbagent/skills/kbagent/references/auth-workflow.md index 3bcdd4eaa..0a3622dd0 100644 --- a/plugins/kbagent/skills/kbagent/references/auth-workflow.md +++ b/plugins/kbagent/skills/kbagent/references/auth-workflow.md @@ -8,7 +8,7 @@ > once, understand what got stored where, and know how to check on / tear > down the session later. > Since v0.80.0 (browser login), v0.84.0 (unattended `login-password`), -> vNEXT (`project create` -- no Keboola account needed at all). +> 0.95.0 (`project create` -- no Keboola account needed at all). > Full command reference: `commands-reference.md` > "Programmatic Auth > (Browser Login)". Gotchas: `gotchas.md` > "Programmatic auth (browser > login) needs a human to approve; sentinel tokens; session scope" and > "`auth @@ -81,7 +81,7 @@ path is unchanged by either feature. ## No Keboola account at all: `project create` -*(since vNEXT, DMD-1940)* +*(since 0.95.0, DMD-1940)* Everything else in this file assumes the user already has a Keboola account. `kbagent project create --url URL` is the one path that does not: it diff --git a/plugins/kbagent/skills/kbagent/references/commands-reference.md b/plugins/kbagent/skills/kbagent/references/commands-reference.md index 73e50b914..b22c175a7 100644 --- a/plugins/kbagent/skills/kbagent/references/commands-reference.md +++ b/plugins/kbagent/skills/kbagent/references/commands-reference.md @@ -72,7 +72,7 @@ fallback, and its message is truncated. See `auth-workflow.md` for the end-to-end walkthrough and troubleshooting. ## Project Management -- `project create --url URL [--project ALIAS] [--name NAME] [--backend snowflake|bigquery] [--sync-backend-init]` -- create a **brand-new** Keboola project from a machine with no Keboola identity: no account, no token, no `auth login` first. The only kbagent command that works from nothing. In one call it provisions the project, stores the returned project-pinned session in `auth.json`, and registers the project in `config.json` under the `kbc-session://` sentinel -- becoming the default project when nothing else was registered, so the next command needs no `--project`. **The project it creates is owned by nobody**: the result's `confirm_url` is a single-use, days-limited link a human must open and sign in at to take ownership -- relay it verbatim, an unclaimed project is a billable orphan. `auth status` re-prints it as `agent_confirm_url` while it is pending. Confirming **revokes this session**, so the next step is always `kbagent auth login --stack URL`; the alias survives it, because the sentinel keys on project id + stack and not on the session. Requires the `agent-provisioning` stack feature (`STACK_FEATURES__AGENT_PROVISIONING`), which is off on most stacks -- the command is always registered, so seeing it in `--help` proves nothing; without the feature the stack 404s and the command exits 1 with `AUTH_NOT_SUPPORTED_ON_STACK`, naming the flag and how to connect an existing project instead. Refuses with exit 5 when a session for that stack already exists (one session per stack; whoever has one has an account and should use the Keboola UI). `--backend` omitted keeps the stack maintainer's default; `--sync-backend-init` waits for backend init instead of the async default, which warns that the first Storage command may fail until it lands (since vNEXT) +- `project create --url URL [--project ALIAS] [--name NAME] [--backend snowflake|bigquery] [--sync-backend-init]` -- create a **brand-new** Keboola project from a machine with no Keboola identity: no account, no token, no `auth login` first. The only kbagent command that works from nothing. In one call it provisions the project, stores the returned project-pinned session in `auth.json`, and registers the project in `config.json` under the `kbc-session://` sentinel -- becoming the default project when nothing else was registered, so the next command needs no `--project`. **The project it creates is owned by nobody**: the result's `confirm_url` is a single-use, days-limited link a human must open and sign in at to take ownership -- relay it verbatim, an unclaimed project is a billable orphan. `auth status` re-prints it as `agent_confirm_url` while it is pending. Confirming **revokes this session**, so the next step is always `kbagent auth login --stack URL`; the alias survives it, because the sentinel keys on project id + stack and not on the session. Requires the `agent-provisioning` stack feature (`STACK_FEATURES__AGENT_PROVISIONING`), which is off on most stacks -- the command is always registered, so seeing it in `--help` proves nothing; without the feature the stack 404s and the command exits 1 with `AUTH_NOT_SUPPORTED_ON_STACK`, naming the flag and how to connect an existing project instead. Refuses with exit 5 when a session for that stack already exists (one session per stack; whoever has one has an account and should use the Keboola UI). `--backend` omitted keeps the stack maintainer's default; `--sync-backend-init` waits for backend init instead of the async default, which warns that the first Storage command may fail until it lands (since 0.95.0) - `project add --project NAME --url URL --token TOKEN` -- connect a project (token verified via API) - `project list` -- list all connected projects (tokens masked). Carries `auth_mode` on every `--json` entry and an `Auth` column (before `Token`) in the human table: exactly `session` (browser login) or `static` (Storage token), **always present and never empty** -- including on a row synthesized for a project whose connection failed -- so branch on it instead of testing for absence. This is how you answer "which mode is this project in": `kbagent project list --json | jq '.data[].auth_mode'`. In the human table a session project's `Token` cell is a dash, because the stored sentinel is not a credential and masked (`kbc-...9840`) it reads like a truncated real token; the `--json` `token` value is unchanged (`mask_token(...)`, a stable contract) and the sentinel's body is the project id, which has its own column (since v0.80.0) - `project remove --project NAME` -- disconnect a project @@ -174,7 +174,7 @@ Requires a **super-admin** Manage API token (same kind as `org setup`). Same def - `storage buckets [--project NAME] [--branch ID]` -- list buckets with sharing/linked info (branch-aware) - `storage bucket-detail --project NAME --bucket-id ID [--branch ID]` -- bucket detail with backend-native direct-access paths (branch-aware). Output adapts to backend: Snowflake -> `snowflake_database` / `snowflake_schema` / per-table `snowflake_path` quoted with `"..."`. BigQuery -> `bigquery_dataset` (and `bigquery_project` when surfaced via API `databaseName`) / per-table `bigquery_path` quoted with backticks. Always-present backend-agnostic keys: `sql_dialect` (`"snowflake"` / `"bigquery"`) and per-table `sql_path` -- prefer these in agent code instead of branching on backend yourself - `storage tables [--project NAME ...] [--bucket-id ID] [--branch ID] [--include-usage]` -- list tables across all connected projects in parallel (multi-project by default, same as `storage buckets`); repeat `--project` to target a subset; `--bucket-id` is applied independently per project (missing buckets become per-project errors); `--branch` requires exactly one `--project`. `--include-usage` (0.88.0+) adds `used_by` per table: the configurations naming it in their **storage input/output mapping** only -- a table id inside a transformation's SQL is NOT a reference. Costs one extra component listing per project (not per table), which is the slow call in a big project; unreadable components degrade to an empty `used_by` -- `storage table-detail --project NAME --table-id ID [--branch ID]` -- table detail with columns, types, primary key, row count (branch-aware). Since 0.88.0 (#621) also returns the raw Storage API `definition`: on BigQuery that carries `timePartitioning` / `rangePartitioning` / `clustering` / `requirePartitionFilter` / `partitions[]`, and it is the only way to verify a repartition landed. Human mode prints the layout and a partition COUNT; `--json` passes `definition` through verbatim. Present on every response (untyped tables too), so `null` means the stack omitted the key, not "untyped". Since 0.88.0 (#624) it also RESOLVES column descriptions written by anyone -- the UI, a component, or kbagent -- into `column_details[].description`, with precedence native definition -> `columnMetadata` `KBC.description` -> legacy flat `KBC.column.*` (an alias table falls back to the source table's `columnMetadata`, matching the MCP server). The response always carries `legacy_column_descriptions`, naming the columns still backed by the pre-0.88.0 convention -- human mode warns and points at `storage describe-migrate`. *(since v0.89.0)* Human mode also renders a `Description` column in the Columns table (only when at least one column has one; long text wraps rather than truncating); on 0.88.0 descriptions were visible in `--json` only. Reading never writes, so it is safe under a read-only token or `--deny-writes`. *(since vNEXT, #761)* The response also carries `backend_path` (the owning bucket's Storage `backendPath`, verbatim: Snowflake `[database, schema]`, BigQuery `[dataset]`) and `sql_path` (the quoted, directly queryable table path, same convention as `storage bucket-detail`'s per-table `sql_path`; `null` when Storage reports no location or the backend is not Snowflake/BigQuery) +- `storage table-detail --project NAME --table-id ID [--branch ID]` -- table detail with columns, types, primary key, row count (branch-aware). Since 0.88.0 (#621) also returns the raw Storage API `definition`: on BigQuery that carries `timePartitioning` / `rangePartitioning` / `clustering` / `requirePartitionFilter` / `partitions[]`, and it is the only way to verify a repartition landed. Human mode prints the layout and a partition COUNT; `--json` passes `definition` through verbatim. Present on every response (untyped tables too), so `null` means the stack omitted the key, not "untyped". Since 0.88.0 (#624) it also RESOLVES column descriptions written by anyone -- the UI, a component, or kbagent -- into `column_details[].description`, with precedence native definition -> `columnMetadata` `KBC.description` -> legacy flat `KBC.column.*` (an alias table falls back to the source table's `columnMetadata`, matching the MCP server). The response always carries `legacy_column_descriptions`, naming the columns still backed by the pre-0.88.0 convention -- human mode warns and points at `storage describe-migrate`. *(since v0.89.0)* Human mode also renders a `Description` column in the Columns table (only when at least one column has one; long text wraps rather than truncating); on 0.88.0 descriptions were visible in `--json` only. Reading never writes, so it is safe under a read-only token or `--deny-writes`. *(since 0.95.0, #761)* The response also carries `backend_path` (the owning bucket's Storage `backendPath`, verbatim: Snowflake `[database, schema]`, BigQuery `[dataset]`) and `sql_path` (the quoted, directly queryable table path, same convention as `storage bucket-detail`'s per-table `sql_path`; `null` when Storage reports no location or the backend is not Snowflake/BigQuery) - `storage create-bucket --project NAME --stage STAGE --name NAME [--description D] [--backend B] [--branch ID]` -- create bucket (branch-aware). With `--branch ID` on a project lacking the `storage-branches` feature (legacy fake-branch), response carries `legacy_branch_storage: true` and human mode prints a warning -- the runner will create a parallel `out.c--*` bucket at job time. See `storage-types-workflow.md` - `storage create-table --project NAME --bucket-id ID --name NAME [--column col:TYPE[(length)] ...] [--primary-key COL] [--not-null COL ...] [--default NAME=VALUE ...] [--source-table-id ID] [--source-branch-id N] [--time-partitioning-type DAY|HOUR|MONTH|YEAR] [--time-partitioning-field COL] [--time-partitioning-expiration-ms MS] [--range-partitioning-field COL --range-partitioning-start S --range-partitioning-end E --range-partitioning-interval I] [--clustering-field COL ...] [--branch ID] [--if-not-exists]` -- create typed table. Base types `STRING/INTEGER/NUMERIC/FLOAT/BOOLEAN/DATE/TIMESTAMP` plus native backend types with length (`VARCHAR(40)`, `NUMBER(18,2)`, `TIMESTAMP_TZ`, `VARIANT`, etc.) -- type/length validation delegated to the Storage API. `--not-null` marks a column `nullable=false`; `--default NAME=VALUE` sets a DEFAULT expression (booleans must be lowercase `true`/`false`). In a dev branch, the target bucket is auto-materialized if it has not yet been written to there -- response surfaces this via `auto_created_bucket: bool`. On legacy fake-branch projects (no `storage-branches` feature), `legacy_branch_storage: true` flags that the runner will use a separate `out.c--*` bucket at job time. `--if-not-exists` (0.47.0+) turns a duplicate-display-name failure into `action: skipped` when the table really exists at the expected id (safe for parallel workers). Since 0.47.1 the skipped envelope reports the EXISTING table's actual `columns`/`primary_key`/`name`, mirrors the request under `requested_columns`/`requested_primary_key`, and sets `schema_drift: true` when they diverge. **`--source-table-id` (0.66.0+, BigQuery only)** copies an existing table's data into the requested partition/clustering layout instead of building from `--column` (schema derived from source -> `--column`/`--not-null`/`--default` forbidden; the two are mutually exclusive). This is the supported way to repartition a populated BigQuery table -- then promote it with `storage swap-tables`. Partition/clustering flags (`--time-partitioning-*`, `--range-partitioning-*`, `--clustering-field`) also work on a plain `--column` create (BigQuery only); time vs range partitioning are mutually exclusive and range bounds are strings. When any source/partition/clustering flag is used, a one-call backend pre-flight rejects non-BigQuery projects (exit 2) before the create. See `storage-types-workflow.md` - `storage upload-table --project NAME --table-id ID --file PATH [--incremental] [--branch ID]` -- upload CSV (branch-aware) @@ -413,11 +413,11 @@ Manage Keboola metastore models -- datasets, metrics, relationships, constraints - `semantic-layer search-context --project P [--pattern G ...] [--type model|dataset|metric|relationship|constraint|glossary|all] [--limit N]` -- project-wide glob search across semantic-layer entity names. Mirrors the upstream `keboola-mcp-server search_semantic_context` MCP tool so a downstream caller can drop the MCP dependency for the pre-flight "is the model populated?" check. Patterns are case-sensitive `fnmatch`, repeatable (union); default `*`. Default `--type all` searches every CHILD type (`model` searches semantic models). `--limit N` short-circuits both per-type and outer loops. Envelope: `{project, contexts: [{id, type, name, description, attributes}], total_count}`; the `type` field is the CLI-friendly singular (no `semantic-` prefix). - `semantic-layer schema --project P (--type model|dataset|metric|relationship|constraint|glossary[,TYPE...] | --all)` -- live JSON Schema per semantic object type, fetched from the deployed metastore (never bundled, cannot drift). Exactly one of `--type`/`--all` (usage error otherwise); `--type` takes a comma-separated list, fan-out is parallel. The bare schema endpoint returns only a version LISTING -- the service resolves the `isDefault` version and fetches the real schema (a deliberate improvement over the upstream `get_semantic_schema` MCP tool, which passes the bare listing through). Envelope: `{project, schemas: [{type, schema, schema_version}]}`. - `semantic-layer get-context --project P --context-id ID` -- single-entry fetch by id, irrespective of type. Probes `semantic-model` first then every CHILD type (dataset / metric / relationship / constraint / glossary) until a 200 lands. 404 on any one type is non-terminal; only a full miss raises `NOT_FOUND` (exit 1). Non-404 errors (500, etc.) propagate immediately rather than being swallowed by the next probe. -- `semantic-layer validate --project P [--model M] [--deep]` -- structural validation. Basic mode runs local checks: duplicate names, dangling rel/metric refs, SUM-on-PCT (warning), constraint orphans (metrics in `metrics[]` that no longer exist), severity-suffix mismatches between API `severity` and the 4-band name suffix. `--deep` adds parallel Snowflake column-existence probes via the in-process StorageService: phantom dataset fields, phantom column refs in metric SQL, AGG-on-STRING errors. *(since vNEXT, #761)* `--deep` also warns `FQN_MISMATCH` when a dataset's stored `fqn` differs from the table's Storage location (`storage table-detail` -> `sql_path`) -- models built before vNEXT carry a `"KEBOOLA"` database that exists in no project. Response: `{valid: bool, deep: bool, errors: [{type, item, detail}], warnings: [...]}`. +- `semantic-layer validate --project P [--model M] [--deep]` -- structural validation. Basic mode runs local checks: duplicate names, dangling rel/metric refs, SUM-on-PCT (warning), constraint orphans (metrics in `metrics[]` that no longer exist), severity-suffix mismatches between API `severity` and the 4-band name suffix. `--deep` adds parallel Snowflake column-existence probes via the in-process StorageService: phantom dataset fields, phantom column refs in metric SQL, AGG-on-STRING errors. *(since 0.95.0, #761)* `--deep` also warns `FQN_MISMATCH` when a dataset's stored `fqn` differs from the table's Storage location (`storage table-detail` -> `sql_path`) -- models built before 0.95.0 carry a `"KEBOOLA"` database that exists in no project. Response: `{valid: bool, deep: bool, errors: [{type, item, detail}], warnings: [...]}`. - `semantic-layer export --project P [--model M] [--output PATH]` -- snapshot the model to a self-describing JSON file (default `./sl_export_{model_name}_{YYYYMMDD_HHMMSS}.json`). Schema-versioned for round-trip via `import` / `diff`. - `semantic-layer diff (--project-a A | --file-a P) (--project-b B | --file-b P) [--model-a M] [--model-b M]` -- three-way diff: project<->project, project<->file, file<->file. Mutually exclusive per side: pass exactly one of `--project-a` / `--file-a`, ditto for B. Output groups changes per entity type: `added[] / removed[] / changed[{name, diff_keys[]}]`. - `semantic-layer add metric --project P [--model M] --name N --sql SQL --dataset TABLE_ID [--description D] [--yes]` -- add a metric. `--dataset` is a Storage tableId (e.g. `out.c-foo.fact_orders`) -- the metric's dataset field stores the tableId, not the dataset name. `--yes` skips the dataset-mismatch confirmation. -- `semantic-layer add dataset --project P [--model M] --name N --table-id TABLE_ID [--description D] [--grain G] [--primary-key COL ...] [--deep-fields] [--fqn FQN]` -- add a dataset. *(since vNEXT, #761)* The `fqn` is the table's warehouse location read from Storage (the owning bucket's `backendPath`, i.e. `storage table-detail` -> `sql_path`), so the table must exist; a linked bucket's path names the SOURCE project's database and schema. `--fqn` stores the given value verbatim and skips the lookup. Before vNEXT the fqn hardcoded a `"KEBOOLA"` database that does not resolve. `--primary-key` is repeatable for composite PKs. `--deep-fields` fetches the storage schema and synthesises role-classified `fields[]`: PK_/FK_->`key`, *_DATE/*_DT->`timestamp`, numeric amount/value/rate->`measure`, else `dimension`. +- `semantic-layer add dataset --project P [--model M] --name N --table-id TABLE_ID [--description D] [--grain G] [--primary-key COL ...] [--deep-fields] [--fqn FQN]` -- add a dataset. *(since 0.95.0, #761)* The `fqn` is the table's warehouse location read from Storage (the owning bucket's `backendPath`, i.e. `storage table-detail` -> `sql_path`), so the table must exist; a linked bucket's path names the SOURCE project's database and schema. `--fqn` stores the given value verbatim and skips the lookup. Before 0.95.0 the fqn hardcoded a `"KEBOOLA"` database that does not resolve. `--primary-key` is repeatable for composite PKs. `--deep-fields` fetches the storage schema and synthesises role-classified `fields[]`: PK_/FK_->`key`, *_DATE/*_DT->`timestamp`, numeric amount/value/rate->`measure`, else `dimension`. - `semantic-layer add relationship --project P [--model M] --name N --from TABLE_ID --to TABLE_ID --on EXPR [--type left|inner]` -- add a join relationship. `--type` defaults to `left`. - `semantic-layer add constraint --project P [--model M] --name N --constraint-type T --rule "EXPR" --metrics M1,M2 [--severity error|warning|info]` -- add a constraint. `--constraint-type` is the closed enum `inequality|equality|range|composition|exclusion|temporal|conditional`. `--rule` is a **STRING expression** (e.g. `"value >= 0"`), NEVER a `{bounds: {min, max}}` object (sl-builder docs are wrong -- see [gotchas.md](gotchas.md)). `--metrics` is a comma-separated list of metric names that must already exist in the model. `--severity` defaults to `warning`. Name regex `^[a-z][a-z0-9_]*$`; the 4-band health convention lives in the name suffix `_critical / _warning / _healthy / _review`, distinct from the 3-value API `severity`. - `semantic-layer add glossary --project P [--model M] --term TERM [--definition D]` -- add a glossary term. @@ -433,7 +433,7 @@ Manage Keboola metastore models -- datasets, metrics, relationships, constraints - `semantic-layer remove glossary --project P [--model M] --term TERM [--yes]` -- destructive. Glossary entries aren't referenced by other entities; no orphan-check. - `semantic-layer import --project P --file PATH [--model M] [--types T,T,...] [--dry-run] [--yes] [--overwrite]` -- replay a snapshot. Default: SKIP on conflict (no surprise overwrites). `--overwrite` opts into DELETE+POST for conflicting items. `--types` filters to a subset (`datasets,metrics,relationships,glossary,constraints`). Dependency-ordered push: datasets -> metrics -> relationships -> glossary -> constraints. Response: `imported: {: {created, skipped, overwritten, failed: [{name, reason}]}}`. - `semantic-layer promote --from-project A --to-project B [--from-model M] [--to-model M] [--types T,T,...] [--dry-run] [--yes]` -- cross-project model copy with `modelUUID` rewrite to the target model's UUID. Classifies items NEW / IDENTICAL / CHANGED (deep-equality after stripping `modelUUID` + timestamps). Additive + overwrite only: NEVER deletes target items absent from source. Holds two MetastoreClients in try/finally. Response: per-type counts + `changes[]` with `diff_keys` and `failed[]`. -- `semantic-layer build --project P [--model M] --tables T,T,... [--name N] [--dry-run] [--keep-on-failure] [--output PATH]` -- non-interactive heuristic builder. **AI caveat**: the existing `ai_client` has no arbitrary-JSON endpoint, so `build` falls back to a deterministic heuristic synthesising one dataset + one COUNT(*) metric + one glossary entry per table (FQN = the table's Storage location, *(since vNEXT)* no longer a hardcoded `"KEBOOLA"` database; fields[] role-classified). Response carries `fallback_used: "heuristic"`. The push loop walks all 5 child types in dependency order -- this **fixes** the `sl-build` skill bug where `semantic-constraint` was silently dropped. `--model` omitted creates a new model (default name `kbagent_build_model` or `--name N`). **Rollback on push failure**: every successfully-POSTed child is DELETEd in reverse PUSH_ORDER, and the model itself is DELETEd if we created it during this call. The wrapped `KeboolaApiError` carries `details.rollback={attempted, posted_children, deleted, failed_deletes, model_created_here, model_deleted, model_uuid}` so operators get full diagnostics. Pass `--keep-on-failure` to preserve the partial state for forensic inspection (mirrors `data-app create --keep-on-failure`); the wrapped error then carries `details.rollback.attempted=False, reason='keep_on_failure'`. +- `semantic-layer build --project P [--model M] --tables T,T,... [--name N] [--dry-run] [--keep-on-failure] [--output PATH]` -- non-interactive heuristic builder. **AI caveat**: the existing `ai_client` has no arbitrary-JSON endpoint, so `build` falls back to a deterministic heuristic synthesising one dataset + one COUNT(*) metric + one glossary entry per table (FQN = the table's Storage location, *(since 0.95.0)* no longer a hardcoded `"KEBOOLA"` database; fields[] role-classified). Response carries `fallback_used: "heuristic"`. The push loop walks all 5 child types in dependency order -- this **fixes** the `sl-build` skill bug where `semantic-constraint` was silently dropped. `--model` omitted creates a new model (default name `kbagent_build_model` or `--name N`). **Rollback on push failure**: every successfully-POSTed child is DELETEd in reverse PUSH_ORDER, and the model itself is DELETEd if we created it during this call. The wrapped `KeboolaApiError` carries `details.rollback={attempted, posted_children, deleted, failed_deletes, model_created_here, model_deleted, model_uuid}` so operators get full diagnostics. Pass `--keep-on-failure` to preserve the partial state for forensic inspection (mirrors `data-app create --keep-on-failure`); the wrapped error then carries `details.rollback.attempted=False, reason='keep_on_failure'`. - `semantic-layer token --encrypt --project P --component-id C` -- encrypt the project's storage token for a transformation's `user_properties`. Builds `{"#metastore_token": }` from the project's already-stored Storage API token and delegates to the existing EncryptService. `--encrypt` is currently required; other modes are refused with `USAGE_ERROR` (exit 2). Output (human): the raw envelope ready to paste. JSON: full `{encrypted, component_id, project}`. ### Reference data (dimension members, e.g. a Chart of Accounts) diff --git a/plugins/kbagent/skills/kbagent/references/gotchas.md b/plugins/kbagent/skills/kbagent/references/gotchas.md index b39d8d9f9..cf6fb121e 100644 --- a/plugins/kbagent/skills/kbagent/references/gotchas.md +++ b/plugins/kbagent/skills/kbagent/references/gotchas.md @@ -13,7 +13,7 @@ Versioning convention: ## A semantic-layer dataset `fqn` is the table's real warehouse location, not `"KEBOOLA"` -*(since vNEXT, #761)* +*(since 0.95.0, #761)* - **`semantic-layer add dataset` and `semantic-layer build` used to write `"KEBOOLA"."".""` for every dataset.** No project has a @@ -160,14 +160,29 @@ Versioning convention: counterpart** -- it did not exist when this rule was written and does not fall under it. - **The device-login panel's one-click link is conditional -- relay only what - is actually printed (since 0.92.0).** Alongside the verification URL and - code, the panel also prints a pre-filled `verification_uri_complete` link - ("Or open this link (code pre-filled):") whenever the stack's device - response includes one; a stack that returns an empty - `verificationUriComplete` prints neither that line nor the label. Don't - promise the user a one-click link unconditionally when relaying the panel - -- check what actually came back and relay only those lines. The URL and - code lines are always present regardless. + is actually printed (since 0.92.0).** The panel prints a pre-filled + `verification_uri_complete` link only when the stack's device response + includes one; a stack that returns an empty `verificationUriComplete` + prints neither the link nor its label. Don't promise the user a one-click + link unconditionally when relaying the panel -- check what actually came + back and relay only those lines. The verification URL and the code are + always present regardless. **The panel's order changed in 0.95.0**: the + pre-filled link is now the headline ("Open this link to finish signing in + (code already filled in):") and the type-it-yourself URL is the fallback + under it ("Or enter the code by hand at this URL:"). It used to be the + other way round, with the link last under "Or open this link (code + pre-filled):" -- so a relay that keys on that old label, or that assumes + the first URL in the panel is the manual one, now picks the wrong line. +- **`press c to copy` is absent exactly when an agent is driving (since + 0.95.0).** In a real terminal the device-login panel is followed by + `Press c to copy the link`, and the keypress is read inside the wait + between device-token polls. The option disables itself -- printing + nothing, output byte-for-byte as before -- in `--json` mode, when stdin or + stdout is not a TTY, or when no native clipboard command is installed + (`pbcopy` on macOS, `clip` on Windows, `wl-copy` / `xclip` / `xsel` / + `clip.exe` on Linux and WSL). A background shell is not a TTY, so the + login you drive will not offer it: relay the link itself, never "press c", + and do not wait for a copy that cannot happen. - **PKCE is the default; the device flow is a fallback, not a mode switch.** The CLI tries the browser (PKCE authorization-code) flow first and falls back to the RFC 8628 device flow ONLY on a *pre-exchange* failure: no @@ -361,7 +376,7 @@ Versioning convention: ## `project create` makes a project nobody owns until a human clicks -*(since vNEXT, DMD-1940)* +*(since 0.95.0, DMD-1940)* - **It is the only kbagent command that works from nothing** -- no account, no token, no `auth login` first: `kbagent project create --url URL diff --git a/plugins/kbagent/skills/kbagent/references/semantic-layer-workflow.md b/plugins/kbagent/skills/kbagent/references/semantic-layer-workflow.md index c95266e3c..fd165e76f 100644 --- a/plugins/kbagent/skills/kbagent/references/semantic-layer-workflow.md +++ b/plugins/kbagent/skills/kbagent/references/semantic-layer-workflow.md @@ -92,9 +92,9 @@ filter your jq with these exact values: Snowflake STRING column (error). - `DEEP_FETCH_FAILED` (`--deep` only) -- couldn't fetch the Snowflake schema for a dataset; deep checks for that dataset are skipped (warning). -- `FQN_MISMATCH` (`--deep` only, since vNEXT) -- a dataset's stored `fqn` +- `FQN_MISMATCH` (`--deep` only, since 0.95.0) -- a dataset's stored `fqn` is not the table's Storage location (`storage table-detail` -> - `sql_path`). Every model built before vNEXT hits this: its fqns name a + `sql_path`). Every model built before 0.95.0 hits this: its fqns name a `"KEBOOLA"` database that exists in no project. Repair recipe in [gotchas.md](gotchas.md) (warning -- an fqn set on purpose with `add dataset --fqn` may point elsewhere). diff --git a/pyproject.toml b/pyproject.toml index c1eb7f055..48c93846a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "keboola-cli" -version = "0.94.0" +version = "0.95.0" description = "AI-friendly CLI for managing Keboola projects" readme = "README.md" requires-python = ">=3.12" diff --git a/src/keboola_agent_cli/changelog.py b/src/keboola_agent_cli/changelog.py index 7c471a60d..4c4d30d14 100644 --- a/src/keboola_agent_cli/changelog.py +++ b/src/keboola_agent_cli/changelog.py @@ -24,6 +24,42 @@ # Ordered newest-first. Each value is a list of brief one-line descriptions. CHANGELOG: dict[str, list[str]] = { + "0.95.0": [ + "New (#775): `kbagent project create --url URL` creates a new Keboola project from a " + "machine with no Keboola account and no token (DMD-1940). It stores the session and " + "registers the project, so other commands can use the project at once. Nobody owns the " + "new project until a person opens the `confirm_url` from the result and signs in. " + "`auth status` shows the link again until then. The confirmation revokes the agent " + "session, so the next step is `kbagent auth login --stack URL`. The stack must have the " + "`agent-provisioning` feature. Without it, the command exits with " + "`AUTH_NOT_SUPPORTED_ON_STACK`. The command refuses to run when a session for the stack " + "already exists. It never retries the request, because each successful call creates a " + "new organization and a billable project.", + "New (#772): the `auth login` device-code panel shows `Press c to copy the link`. Press " + "`c` to copy the sign-in link to the clipboard. kbagent uses the native clipboard command (`pbcopy`, `clip`, `wl-copy`, `xclip`, `xsel`, or " + "`clip.exe` on WSL). kbagent does not show the hint in `--json` mode, outside a terminal " + "(TTY), or when no clipboard command is installed.", + "Fix (#762): `semantic-layer add dataset` and `semantic-layer build` now take the dataset " + "`fqn` from the table's real location in Storage (#761). Before, they wrote a " + '`"KEBOOLA"` database that exists in no project, so every consumer that used the `fqn` ' + "in SQL failed (Kai, data apps, AI SQL generation). For a linked bucket, the `fqn` names " + "the database and schema of the source project. `add dataset` has a new `--fqn` option " + "that stores the given value as-is. `storage table-detail` returns two new keys, " + "`backend_path` and `sql_path`. `semantic-layer validate --deep` shows a `FQN_MISMATCH` " + "warning for each dataset whose stored `fqn` differs from the table location. Every " + "model built before this release gets this warning. kbagent does not change existing " + "models. Behavior change: `add dataset` without `--fqn` now fails when the table does " + "not exist. `build` stops before its first write when it cannot find the location of a " + "table.", + "Change (#763): the release pipeline publishes kbagent to WinGet again. The WinGet " + "package `Keboola.KeboolaCLI2` stayed at 0.79.0, because the `winget` job stayed " + "disabled after the first manual submission. 0.95.0 is the first release with the job " + "enabled.", + "Note (#768, #781, #782, #773): housekeeping with no user-facing change. The CI " + "dependency audit uses `uv audit` instead of `pip-audit`, Dependabot can update " + "`uv.lock` again, a dependency update clears an audit finding, and the `_url_copy` tests " + "are type-clean.", + ], "0.94.0": [ "New (#759): a browser-login session (`auth login`) now works with nearly every command, " "not just the Storage and Manage paths (CLI-13). It reaches the Scheduler (`flow schedule` / " diff --git a/src/keboola_agent_cli/commands/context.py b/src/keboola_agent_cli/commands/context.py index f9fe59b19..8ac954e4f 100644 --- a/src/keboola_agent_cli/commands/context.py +++ b/src/keboola_agent_cli/commands/context.py @@ -706,7 +706,7 @@ 0.88.0 column descriptions were readable through --json column_details[].description only. Also returns `backend_path` (the owning bucket's Storage backendPath, verbatim) and `sql_path` (the quoted, directly queryable table path; null when Storage reports no location) (since - vNEXT, #761). A linked bucket's path is the SOURCE project's database + schema. + 0.95.0, #761). A linked bucket's path is the SOURCE project's database + schema. kbagent storage create-bucket --project NAME --stage STAGE --name BUCKET_NAME [--description D] [--backend B] [--branch ID] Create a new storage bucket. Stage must be "in" or "out". Branch-aware. @@ -1708,8 +1708,8 @@ constraint orphans, severity-suffix). --deep adds parallel Snowflake column-existence checks for phantom fields, phantom column refs, and AGG-on-STRING via in-process StorageService, plus an FQN_MISMATCH warning - for a dataset `fqn` that is not the table's Storage location (since vNEXT; - models built before vNEXT carry a nonexistent "KEBOOLA" database). + for a dataset `fqn` that is not the table's Storage location (since 0.95.0; + models built before 0.95.0 carry a nonexistent "KEBOOLA" database). kbagent semantic-layer export --project P [--model M] [--output PATH] Snapshot the model to a self-describing JSON file. Default path: diff --git a/uv.lock b/uv.lock index e6625209f..a6dbbc6e1 100644 --- a/uv.lock +++ b/uv.lock @@ -465,7 +465,7 @@ wheels = [ [[package]] name = "keboola-cli" -version = "0.94.0" +version = "0.95.0" source = { editable = "." } dependencies = [ { name = "croniter" },