diff --git a/CLAUDE.md b/CLAUDE.md index 699962ee9..6de1fa7a3 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -60,7 +60,7 @@ The full repo guide — stack, structure, commands, and conventions — lives in - Conserve content: every removed paragraph → a new home or a logged deletion (CONSERVATION-REPORT + MISSING list). Never drop silently. - Current guardrails: don't assert E2B as hosting; don't say Streamlit is retired; - don't document pricing / in-app Kai-toggle / semantic layer. + don't document pricing / in-app Kai-toggle. - Stakeholder likes/dislikes + the paste-ready prompt block live in the local `apps-section/` bundle (gitignored), not here. diff --git a/_data/cli/command-reference.md b/_data/cli/command-reference.md index df08859ad..5d2696eda 100644 --- a/_data/cli/command-reference.md +++ b/_data/cli/command-reference.md @@ -1,6 +1,6 @@ # kbagent command reference -Generated from kbagent v0.76.1 by `scripts/gen_command_reference.py`. +Generated from kbagent v0.91.0 by `scripts/gen_command_reference.py`. Derived from the CLI's own command tree -- do not edit by hand. ## Global options @@ -58,6 +58,63 @@ Check if a specific operation is allowed. |---|---|---| | `operation` (positional) | yes | | +## `auth` + +Programmatic browser login (PKCE / device code) -- user-scoped sessions + +### `kbagent auth login` + +Sign in to a Keboola stack via browser login (PKCE) or device code. + +| Option | Required | Description | +|---|---|---| +| `--stack` `` | | Stack URL or a registered project alias to log into | +| `--device-code` | | Force the device-authorization flow (skip the browser loopback) | +| `--register-projects` | | Register every project this session can access under a local alias | + +### `kbagent auth login-password` + +Sign in via email + password (+ TOTP if the account has MFA) -- no browser. + +| Option | Required | Description | +|---|---|---| +| `--email` `` | yes | Account email. Also settable via KBC_LOGIN_EMAIL. | +| `--password` `` | | Account password. Prefer KBC_LOGIN_PASSWORD (a CI secret in the step's env: block) or --password-stdin over typing this flag directly -- it avoids the value landing in shell history or a process listing. | +| `--password-stdin` | | Read the password from stdin instead of --password/KBC_LOGIN_PASSWORD. On a TTY this is a hidden prompt (Enter to confirm); on a pipe it reads until EOF (e.g. `echo "$PASS" | kbagent auth login-password --password-stdin ...`). | +| `--totp-secret` `` | | Base32 TOTP seed (from the account's authenticator enrollment, NOT a 6-digit code) -- required if the account has TOTP-based MFA configured. kbagent computes the current code from this itself, so no human ever types a live code. Also settable via KBC_LOGIN_TOTP_SECRET. | +| `--stack` `` | | Stack URL or a registered project alias to log into | +| `--register-projects` | | Register every project this session can access under a local alias | + +### `kbagent auth status` + +Show the programmatic-auth session health for a stack. + +| Option | Required | Description | +|---|---|---| +| `--stack` `` | | Stack URL or a registered project alias to inspect | + +### `kbagent auth logout` + +Revoke and clear the local programmatic-auth session for a stack. + +| Option | Required | Description | +|---|---|---| +| `--stack` `` | | Stack URL or a registered project alias to log out of | +| `--remove-projects` | | Also remove local project aliases registered from this session | +| `--yes` / `-y` | | Skip confirmation prompt | + +### `kbagent auth register-projects` + +Register accessible projects from the current session as local aliases. + +| Option | Required | Description | +|---|---|---| +| `--stack` `` | | Stack URL or a registered project alias | +| `--all` | | Register every accessible project. Mutually exclusive with --project-id. | +| `--project-id` `` | | Register only this project id (repeatable). Mutually exclusive with --all. | +| `--alias` `` | | Alias override as ID=ALIAS (repeatable). Applies in every mode, including as the prefilled default inside the interactive picker. | +| `--yes` / `-y` | | Skip the picker's final confirmation prompt | + ## `project` Manage connected Keboola projects @@ -92,7 +149,7 @@ Edit an existing Keboola project connection. |---|---|---| | `--project` `` | yes | Alias of the project to edit | | `--url` `` | | New Keboola stack URL | -| `--token` `` | | New Storage API token | +| `--token` `` | | New Storage API token. On a project registered by 'kbagent auth login' this deliberately converts it to a static-token project, which 'kbagent auth logout --remove-projects' no longer cleans up; a warning says so. | | `--new-alias` `` | | Rename the project alias. Updates the config.json projects key AND the default_project field if it matched. Renames the nested sync directory // when present (with -2-suffix collision handling). Lineage cache (if any) is NOT auto-updated; rebuild with 'kbagent lineage build' after the rename. | | `--dry-run` | | Preview the edit without mutating state. Validates --new-alias, detects collision against existing projects, predicts the disk-rename method (git_mv vs shutil_move), and surfaces the lineage-cache warning if any -- all read-only. Errors (collision, invalid format) raise the same exit codes as the live path. No API call is made for --token in dry-run mode. | @@ -116,7 +173,7 @@ Refresh expired or invalid Storage API tokens. | `--force` | | Refresh even if token is valid | | `--yes` / `-y` | | Skip confirmation prompt | | `--token-description` `` | | Description prefix for created Storage API tokens | -| `--token-expires-in` `` | | Token lifetime in seconds. If not set, tokens never expire. | +| `--token-expires-in` `` | | Token lifetime in seconds. If not set, tokens never expire. | ### `kbagent project use` @@ -169,7 +226,7 @@ Invite a user (or many users via CSV) to one or more projects. | `--reason` `` | | Optional human-readable reason attached to the invitation | | `--from-csv` `` | | CSV file with columns email, project (alias or numeric ID), role[, reason] | | `--default-role` `` | | Role to apply when a CSV row has no role column | -| `--workers` `` | | Parallel workers for --from-csv (default 8) | +| `--workers` `` | | Parallel workers for --from-csv (default 8) | | `--dry-run` | | Preview without sending invitations | ### `kbagent project member-list` @@ -236,7 +293,7 @@ Set up projects and register them in the kbagent config. | `--dry-run` | | Preview what would happen without making changes | | `--yes` / `-y` | | Skip confirmation prompt | | `--token-description` `` | | Description prefix for created Storage API tokens | -| `--token-expires-in` `` | | Token lifetime in seconds (e.g. 3600 for 1 hour). If not set, tokens never expire. | +| `--token-expires-in` `` | | Token lifetime in seconds (e.g. 3600 for 1 hour). If not set, tokens never expire. | | `--refresh` | | Refresh tokens for already-registered projects with invalid tokens | ## `feature` @@ -332,6 +389,16 @@ Mint a scoped Storage API token (secret shown once). | `--can-read-all-file-uploads` | | Allow reading files uploaded by OTHER tokens (default: only its own) | | `--expires-in` `` | | Lifetime in seconds (omit = never expires) | +### `kbagent token list` + +List the project's Storage API tokens (no secrets -- those are mint-only). + +| Option | Required | Description | +|---|---|---| +| `--project` / `-p` `` | yes | Project alias | +| `--with-last-used` | | Derive each token's last activity (one extra API call PER TOKEN) and sort dormant-first | +| `--columns` `` | | Table columns to show, in order (repeat for multiple). Available: id, description, created, refreshed, expires, master, created_by, last_used, last_used_event. Human output only -- --json is unaffected. | + ### `kbagent token delete` Revoke a Storage API token immediately (destructive; only non-master tokens). @@ -352,6 +419,18 @@ Rotate a token: generate a new value and invalidate the old one (secret shown on | `--token-id` `` | yes | ID of the token to rotate | | `--yes` / `-y` | | Skip confirmation prompt | +## `billing` + +PAYG credit balance across projects (issue #594). + +### `kbagent billing credits` + +Show the current PAYG credit balance for one or more projects. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | | Project alias (repeatable; omit for all registered projects) | + ## `component` Discover and inspect Keboola components @@ -457,6 +536,7 @@ Update a configuration's metadata and/or content. | `--configuration-file` `` | | Path to a JSON file with configuration content | | `--set` `` | | Set a nested value: PATH VALUE (e.g. --set 'parameters.db.host=new-host') | | `--merge` | | Deep-merge into existing config instead of replacing | +| `--change-description` `` | | Version changeDescription for the audit trail (default: auto-generated) | | `--dry-run` | | Show what would change without applying | | `--branch` `` | | Update in a specific dev branch ID (defaults to active branch) | | `--allow-plaintext-on-encrypt-failure` | | Allow write even if secret encryption fails (DANGEROUS: secrets stored as plaintext) | @@ -490,7 +570,7 @@ Rename a configuration (update name via API + rename local sync directory). ### `kbagent config delete` -Delete a configuration from a project. +Soft-delete a configuration into the trash (restorable). | Option | Required | Description | |---|---|---| @@ -498,6 +578,7 @@ Delete a configuration from a project. | `--component-id` `` | yes | Component ID (e.g. keboola.python-transformation-v2) | | `--config-id` `` | yes | Configuration ID to delete | | `--branch` `` | | Delete from a specific dev branch ID (defaults to active branch) | +| `--dry-run` | | Report what would happen (live / already in trash) without deleting | ### `kbagent config new` @@ -651,6 +732,7 @@ Update an existing configuration row. | `--configuration` `` | | Row configuration JSON: inline, @file.json, or - for stdin | | `--set` `` | | Set a nested value: PATH=VALUE (e.g. --set 'parameters.table=orders') | | `--merge` | | Deep-merge into existing row config instead of replacing | +| `--change-description` `` | | Version changeDescription for the audit trail (default: auto-generated) | | `--dry-run` | | Show what would change without applying | | `--is-disabled` | | Disable the row (mutually exclusive with --is-enabled) | | `--is-enabled` | | Enable the row (mutually exclusive with --is-disabled) | @@ -670,6 +752,52 @@ Delete a configuration row. | `--branch` `` | | Delete from a specific dev branch ID (defaults to active branch) | | `--yes` / `-y` | | Skip confirmation prompt | +### `kbagent config state-get` + +Read the runtime ``state`` dict of a configuration or one of its rows. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--component-id` `` | yes | Component ID | +| `--config-id` `` | yes | Configuration ID | +| `--row-id` `` | | Read this row's state instead of the config's root state | +| `--branch` `` | | Dev branch ID (defaults to active branch) | + +### `kbagent config state-set` + +Overwrite the runtime ``state`` dict of a configuration or one of its rows. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--component-id` `` | yes | Component ID | +| `--config-id` `` | yes | Configuration ID | +| `--state` `` | yes | New state as inline JSON object, @file, or - for stdin | +| `--row-id` `` | | Write this row's state instead of the config's root state | +| `--branch` `` | | Dev branch ID (defaults to active branch) | +| `--dry-run` | | Show what would change without applying | +| `--yes` / `-y` | | Skip confirmation prompt | + +### `kbagent config clone` + +Duplicate a configuration, whole -- including runtime, storage and authorization. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Source project alias | +| `--component-id` `` | yes | Component ID (e.g. keboola.wr-db-snowflake) | +| `--config-id` `` | yes | Configuration ID to clone | +| `--name` `` | yes | Name for the new configuration | +| `--target-project` `` | | Clone into a different project (default: same project) | +| `--description` `` | | Description for the clone (default: inherit the source's) | +| `--set` `` | | Override a value in the clone: PATH=VALUE (repeatable) | +| `--secret` `` | | Re-supply an encrypted value for a cross-project clone: PATH=VALUE (repeatable). Encrypted in the TARGET project on write. | +| `--branch` `` | | Source dev branch (defaults to the active branch) | +| `--target-branch` `` | | Target dev branch (defaults to the target's active branch) | +| `--dry-run` | | Show the plan (and any missing secrets) without writing | +| `--allow-plaintext-on-encrypt-failure` | | Allow the clone even if secret encryption fails (DANGEROUS: plaintext secrets) | + ### `kbagent config oauth-url` Requires master token. @@ -681,6 +809,27 @@ Requires master token. | `--config-id` `` | yes | Configuration ID to authorize | | `--redirect-url` `` | | Optional URL to return to after the OAuth flow completes (sets returnUrl query param) | +### `kbagent config restore` + +Restore a configuration from the trash (undo of 'config delete'). + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--component-id` `` | yes | Component ID (e.g. keboola.snowflake-transformation) | +| `--config-id` `` | yes | Trashed configuration ID | +| `--branch` `` | | Restore in a specific dev branch ID (defaults to active branch) | + +### `kbagent config trash-list` + +List configurations in the trash (restorable via 'config restore'). + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--component-id` `` | | Limit to one component's trashed configurations | +| `--branch` `` | | List a specific dev branch's trash (defaults to active branch) | + ## `data-app` Keboola data-app lifecycle (create, deploy, manage) @@ -727,6 +876,7 @@ Create a Keboola data app end-to-end (POST + encrypt + PUT + deploy). | `--size` `` | | Runtime size: tiny, small, medium, or large. | | `--auto-suspend` `` | | Auto-suspend after N seconds idle (0 disables). | | `--type` `` | | Runtime type. Default 'python-js' covers Python AND Node apps. | +| `--workspace` / `--no-workspace` | | Grant the app Storage access by writing runtime.workspace.enabled=true (default). This is what makes the platform inject WORKSPACE_ID / QUERY_SERVICE_URL / KBC_WORKSPACE_MANIFEST_PATH -- without it an app that reads Storage deploys and reports running while serving no data. Pass --no-workspace only for an app that never touches Storage. | | `--branch` `` | | Keboola dev branch ID (defaults to production). | | `--no-deploy` | | Skip the deploy step; create the shell + Storage config only. | | `--wait` | | Block until state == running (or error). Respects pitfall #1: stopped is not terminal. | @@ -921,6 +1071,9 @@ List jobs from connected projects. | `--config-id` `` | | Filter by configuration ID (requires --component-id) | | `--status` `` | | Filter by job status: processing, terminated, cancelled, success, error | | `--limit` `` | | Maximum number of jobs to return per project (1-500) | +| `--offset` `` | | Number of jobs to skip per project, for paging past --limit | +| `--sort-by` `` | | Field to sort by: startTime, endTime, createdTime, durationSeconds, id | +| `--sort-order` `` | | Sort direction: asc, desc | ### `kbagent job detail` @@ -930,6 +1083,7 @@ Show detailed information about a specific job. |---|---|---| | `--project` `` | yes | Project alias | | `--job-id` `` | yes | Job ID | +| `--log-tail-lines` `` | | Also fetch this many of the job's most recent events (0 = skip the extra call) | ### `kbagent job run` @@ -1000,10 +1154,11 @@ List storage tables from one or more projects. | `--project` `` | | Project alias (can be repeated for multiple projects). Omit to query all connected projects in parallel. | | `--bucket-id` `` | | Filter tables by bucket ID (applied independently per project) | | `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | +| `--include-usage` | | Also report which configurations read or write each table, from their storage input/output mapping. Costs one extra component listing per project (not per table), but that listing carries every configuration body -- expect it to be slow in a big project. | ### `kbagent storage table-detail` -Show detailed table info including columns and types. +Show detailed table info including columns, types and physical layout. | Option | Required | Description | |---|---|---| @@ -1173,53 +1328,6 @@ Delete one or more storage buckets. | `--yes` / `-y` | | Skip confirmation prompt | | `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | -### `kbagent storage describe-bucket` - -Set the description on a storage bucket. - -| Option | Required | Description | -|---|---|---| -| `--project` `` | yes | Project alias | -| `--bucket-id` `` | yes | Bucket ID (e.g. 'in.c-my-bucket') | -| `--text` `` | | Description text (inline) | -| `--file` `` | | Path to a file containing the description | -| `--stdin` | | Read description from standard input | -| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | - -### `kbagent storage describe-table` - -Set the description on a storage table. - -| Option | Required | Description | -|---|---|---| -| `--project` `` | yes | Project alias | -| `--table-id` `` | yes | Table ID (e.g. 'in.c-my-bucket.my-table') | -| `--text` `` | | Description text (inline) | -| `--file` `` | | Path to a file containing the description | -| `--stdin` | | Read description from standard input | -| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | - -### `kbagent storage describe-column` - -Set descriptions on one or more columns of a storage table. - -| Option | Required | Description | -|---|---|---| -| `--project` `` | yes | Project alias | -| `--table-id` `` | yes | Table ID (e.g. 'in.c-my-bucket.my-table') | -| `--column` `` | yes | Column description as 'NAME=DESCRIPTION' (can be repeated) | -| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | - -### `kbagent storage describe-batch` - -Apply descriptions to buckets, tables, and columns from a YAML file. - -| Option | Required | Description | -|---|---|---| -| `--project` `` | yes | Project alias | -| `--from-file` `` | yes | Path to a YAML file with bucket/table/column descriptions | -| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | - ### `kbagent storage files` List Storage Files with optional tag filtering. @@ -1374,6 +1482,67 @@ Create a NEW table from an existing snapshot (snapshot restore). | `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | | `--dry-run` | | Show what would be created without executing | +### `kbagent storage describe-bucket` + +Set the description on a storage bucket. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--bucket-id` `` | yes | Bucket ID (e.g. 'in.c-my-bucket') | +| `--text` `` | | Description text (inline) | +| `--file` `` | | Path to a file containing the description | +| `--stdin` | | Read description from standard input | +| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | + +### `kbagent storage describe-table` + +Set the description on a storage table. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--table-id` `` | yes | Table ID (e.g. 'in.c-my-bucket.my-table') | +| `--text` `` | | Description text (inline) | +| `--file` `` | | Path to a file containing the description | +| `--stdin` | | Read description from standard input | +| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | + +### `kbagent storage describe-column` + +Set descriptions on one or more columns of a storage table. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--table-id` `` | yes | Table ID (e.g. 'in.c-my-bucket.my-table') | +| `--column` `` | yes | Column description as 'NAME=DESCRIPTION' (can be repeated) | +| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | + +### `kbagent storage describe-batch` + +Apply descriptions to buckets, tables, and columns from a YAML file. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--from-file` `` | yes | Path to a YAML file with bucket/table/column descriptions | +| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | + +### `kbagent storage describe-migrate` + +Convert legacy KBC.column.* descriptions to the native definition endpoint. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--table-id` `` | | Migrate only this table (can be repeated). Excludes --bucket-id. | +| `--bucket-id` `` | | Migrate every table of this bucket. Excludes --table-id. | +| `--prune-orphans` | | Also delete legacy entries for columns that no longer exist | +| `--dry-run` | | Show what would be migrated without writing | +| `--yes` / `-y` | | Skip confirmation prompt | +| `--branch` `` | | Dev branch ID (defaults to active branch if set via 'branch use') | + ## `stream` Data Streams (OTLP) source management @@ -1469,6 +1638,7 @@ Link a shared bucket into a project. | `--source-project-id` `` | yes | ID of the project that owns the shared bucket. | | `--bucket-id` `` | yes | Source bucket ID to link (e.g. out.c-data). | | `--name` `` | | Display name for the linked bucket. Auto-generated if omitted. | +| `--stage` `` | | Stage for the linked bucket in the target project: 'in' or 'out'. | ### `kbagent sharing unlink` @@ -1800,6 +1970,67 @@ Audit schedules by cron window or job-freshness. | `--not-run-since` `` | | Only include schedules whose parent config has not produced a job in the last N days (or never ran). Pass 0 to force the last_run_at lookup for every row without applying a staleness filter. NOTE: Queue API is not branch-aware -- combining with --branch still compares against production jobs. | | `--branch` `` | | Dev branch ID (requires single --project) | +## `notification` + +Manage notification subscriptions across projects (Flow Notifications tab). + +### `kbagent notification list` + +List notification subscriptions (Flow Notifications tab) across projects. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | | Project alias (repeatable; omit for all registered projects) | +| `--event` `` | | Event name filter, e.g. 'job-failed'. Known events: job-failed, job-succeeded, job-succeeded-with-warning, job-processing-long, phase-job-failed, phase-job-succeeded, phase-job-succeeded-with-warning, phase-job-processing-long. Not validated against that list -- the service may add more. | +| `--component-id` `` | | Only subscriptions filtering on this component (e.g. keboola.flow) | +| `--config-id` `` | | Only subscriptions filtering on this configuration ID | + +### `kbagent notification detail` + +Show one notification subscription, including its raw filter list. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--subscription-id` `` | yes | Notification subscription ID | + +### `kbagent notification create` + +Create a notification subscription. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--event` `` | yes | Event name filter, e.g. 'job-failed'. Known events: job-failed, job-succeeded, job-succeeded-with-warning, job-processing-long, phase-job-failed, phase-job-succeeded, phase-job-succeeded-with-warning, phase-job-processing-long. Not validated against that list -- the service may add more. | +| `--channel` `` | yes | Recipient channel: email | webhook | +| `--address` `` | yes | Email address (channel=email) or callback URL (channel=webhook) | +| `--component-id` `` | | Restrict to jobs of this component (e.g. keboola.flow) | +| `--config-id` `` | | Restrict to jobs of this configuration ID | +| `--branch` `` | | Restrict to jobs on this branch | +| `--expires-at` `` | | Optional ISO-8601 expiry for the subscription | + +### `kbagent notification delete` + +Delete a notification subscription. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--subscription-id` `` | yes | Notification subscription ID | +| `--yes` / `-y` | | Skip confirmation prompt | + +### `kbagent notification replace-recipient` + +Replace a subscription's recipient. + +| Option | Required | Description | +|---|---|---| +| `--project` `` | yes | Project alias | +| `--subscription-id` `` | yes | Notification subscription ID to replace | +| `--address` `` | yes | New email address or webhook URL for the recipient | +| `--channel` `` | | Override the recipient channel: email | webhook. Defaults to the old subscription's channel. | +| `--yes` / `-y` | | Skip confirmation prompt | + ## `branch` Manage development branches @@ -1965,6 +2196,9 @@ Load tables into a workspace. | `--workspace-id` `` | yes | Workspace ID | | `--tables` `` | yes | Table ID to load (can be repeated, e.g. in.c-bucket.table-name) | | `--preserve` | | Keep existing tables in the workspace (default: clear before loading) | +| `--load-type` `` | | clone|copy|view. Default (omitted) is auto: a zero-copy CLONE for every table the workspace backend can clone, COPY for the rest. An explicit value is sent as-is; an ineligible combination is rejected by the API with the exact reason. | +| `--force` | | Skip the size guard that asks before COPYing a table larger than 1 GB. | +| `--timeout` `` | | Seconds to wait for the load job. On timeout the job KEEPS RUNNING server-side -- kbagent stops watching, it does not cancel. | ### `kbagent workspace query` @@ -1978,7 +2212,7 @@ Execute SQL query in a workspace via Query Service. | `--file` `` | | Path to a .sql file to execute | | `--transactional` | | Wrap query in a transaction | | `--full` | | Fetch the complete result set via CSV export (slower). Default fetches a fast inline page capped by --limit. | -| `--limit` `` | | Max rows to fetch via the fast inline path (ignored with --full). | +| `--limit` `` | | Max rows to fetch via the fast inline path (ignored with --full). | ### `kbagent workspace gc` @@ -2002,30 +2236,6 @@ Create a workspace from a transformation config. | `--row-id` `` | | Optional row ID for row-based transformations | | `--backend` `` | | Workspace backend (auto-detected from project if omitted) | -## `tool` - -MCP tools - interact with Keboola via MCP server - -### `kbagent tool list` - -List available MCP tools from the keboola-mcp-server. - -| Option | Required | Description | -|---|---|---| -| `--project` `` | | Project alias to query tools from (uses first available if not set) | -| `--branch` `` | | Development branch ID (requires --project or active branch) | - -### `kbagent tool call` - -Call an MCP tool on keboola-mcp-server. - -| Option | Required | Description | -|---|---|---| -| `tool_name` (positional) | yes | | -| `--project` `` | | Project alias (required for write tools, optional for read tools) | -| `--input` `` | | Tool input as JSON string, @file.json, or - for stdin | -| `--branch` `` | | Development branch ID (forces single-project mode) | - ## `sync` Sync project configurations with local filesystem @@ -2618,16 +2828,12 @@ Register a new scheduled task. | `--cron` `` | | Cron expression (UTC) | | `--manual` | | Skip cron firing -- only run when triggered manually or as downstream. | | `--enabled` / `--disabled` | | Initial enabled state. | -| `--type` `` | | Action type when not using --from-file: ai_agent|cli_command|mcp_tool | +| `--type` `` | | Action type when not using --from-file: ai_agent|cli_command | | `--from-file` `` | | Full action JSON ({"type": "...", "params": {...}}). PATH, @path, or - for stdin. | | `--cli` `` | | ai_agent: claude|codex|gemini | | `--prompt` `` | | ai_agent: prompt body | | `--extra-arg` `` | | ai_agent: extra CLI arg (repeatable). Forwarded to claude/codex/gemini. | | `--argv` `` | | cli_command: argv element (repeatable). 'kbagent' prefix is auto-added. | -| `--tool` `` | | mcp_tool: tool name (e.g. get_jobs) | -| `--mcp-project` `` | | mcp_tool: project alias to dispatch into. | -| `--mcp-branch` `` | | mcp_tool: branch ID (optional). | -| `--input` `` | | mcp_tool: JSON input. Inline, @path, or -. | | `--timeout` `` | | Action timeout in seconds. | | `--trigger-task-id` `` | | Chain: ID of downstream task to fire after this one. | | `--trigger-on` `` | | Chain filter: success|error|always. | @@ -2711,16 +2917,12 @@ Execute an action ad-hoc (no persistence, no scheduling). |---|---|---| | `--name` `` | | Name shown in event init payload. | | `--stream` | | Stream events live instead of returning the final run record. | -| `--type` `` | | ai_agent|cli_command|mcp_tool | +| `--type` `` | | ai_agent|cli_command | | `--from-file` `` | | Action JSON (or @path / -). | | `--cli` `` | | | | `--prompt` `` | | | | `--extra-arg` `` | | | | `--argv` `` | | | -| `--tool` `` | | | -| `--mcp-project` `` | | | -| `--mcp-branch` `` | | | -| `--input` `` | | | | `--timeout` `` | | | ### `kbagent agent cron-preview` @@ -2893,19 +3095,15 @@ Initialize a local .kbagent/ workspace in the current directory. |---|---|---| | `--from-global` | | Copy projects from the global config into the new local workspace. | | `--project` `` | | Copy only the named project(s) from the global config (repeatable). Implies --from-global. Without it, all global projects are copied. | -| `--read-only` | | Set read-only permission policy (blocks all write CLI commands and MCP tools). | +| `--read-only` | | Set read-only permission policy (blocks all write CLI commands). | ### `kbagent doctor` Run health checks on CLI configuration and project connectivity. -| Option | Required | Description | -|---|---|---| -| `--fix` | | Auto-fix issues: install MCP server binary for faster startup. | - ### `kbagent version` -Show kbagent version and check for dependency updates. +Show the kbagent version and check for updates. | Option | Required | Description | |---|---|---| @@ -2913,7 +3111,7 @@ Show kbagent version and check for dependency updates. ### `kbagent update` -Update kbagent + keboola-mcp-server to the latest versions. +Update kbagent to the latest version. | Option | Required | Description | |---|---|---| @@ -2925,7 +3123,7 @@ Show recent changelog (what changed in each version). | Option | Required | Description | |---|---|---| -| `--limit` / `-n` `` | | Number of versions to show. | +| `--limit` / `-n` `` | | Number of versions to show. | | `--full` / `-v` | | Show complete notes for each version (default: one-line summary). | ### `kbagent context` @@ -2947,9 +3145,10 @@ Launch the kbagent HTTP API server. | `--reload` | | Auto-reload on code changes (uvicorn --reload). | | `--log-level` `` | | uvicorn log level: critical, error, warning, info, debug, trace. | | `--cors-origin` `` | | Add a CORS origin (repeatable). Default: localhost:5173 / 8000. | -| `--config-dir` `` | | Override config directory path (matches kbagent --config-dir). | +| `--config-dir` `` | | Config directory to serve. Wins over a root-level `kbagent --config-dir`; when neither is given, the server falls back to KBAGENT_CONFIG_DIR, then the .kbagent walk-up, then global. | | `--ui` | | Mount the built React SPA at / so a single uvicorn process serves both the API and the UI. ``GET /`` sets an HttpOnly `kbagent_session` cookie (SameSite=Strict, Path=/) so the browser boots already authenticated -- no Node BFF, no paste step, no token in the JS heap or URL. Run `make web-build` once to produce the dist/ folder. | | `--ui-dist` `` | | Override the path to the built React dist/ directory. Defaults to /web/frontend/dist relative to the package, or $KBAGENT_UI_DIST. Implies --ui. | +| `--no-banner` | | Suppress the What's-new popup in the web UI. | ### `kbagent search` @@ -2960,6 +3159,7 @@ Search for items (tables, buckets, configs, flows, …) by name or content. | `query` (positional) | yes | | | `--project` / `-p` `` | | Project alias to search (repeatable; defaults to all projects). | | `--type` / `-t` `` | | Item type to restrict results. Repeatable. Valid values: table, bucket, config, flow, data-app, transformation. | -| `--search-type` `` | | Search mode. ``textual`` (default) searches item names via the Storage API. ``config-based`` scans full configuration JSON bodies. | -| `--limit` / `-l` `` | | Maximum number of results per project (textual search only, 1-100). | +| `--search-type` `` | | Search mode. ``textual`` (default) searches item names via the Storage API. ``config-based`` scans full configuration JSON bodies. Both modes match case-insensitively. | +| `--limit` / `-l` `` | | Maximum number of results per project (textual search only, 1-100). | | `--regex` / `-r` | | Run the query as a regular expression (opt-in). Case-insensitive whole-term match against entity names only (not column names): 'report' will NOT match 'monthly_report' -- use '.*report.*'. Textual search only. | +| `--scope` `` | | Narrow a config-based hit to part of the configuration body. Dot notation written as it appears in the configuration, e.g. 'parameters' or 'storage.input'. Repeatable (scopes are OR-ed). A configuration with no in-scope match drops out of the results. Config-based search only. | diff --git a/_data/navigation.yml b/_data/navigation.yml index 8cdbaa235..f8550867d 100644 --- a/_data/navigation.yml +++ b/_data/navigation.yml @@ -50,41 +50,6 @@ items: - url: /getting-started/next-steps/ title: Where to Go Next - - title: Going Further - items: - - url: /getting-started/load/googlesheets/ - title: Load from Google Sheets - - - url: /getting-started/load/database/ - title: Load from a Database - - - url: /getting-started/transform/workspace/ - title: Use a Workspace - - - url: /getting-started/ad-hoc/ - title: Ad-Hoc Data Analysis - - - url: /getting-started/branches/ - title: Development Branches - items: - - url: /getting-started/branches/prepare-tables/ - title: Prepare Tables - - - url: /getting-started/branches/prepare-files/ - title: Prepare Files - - - title: Tables in Branch - url: /getting-started/branches/tables-in-branch/ - - - title: Files in Branch - url: /getting-started/branches/files-in-branch/ - - - url: /getting-started/branches/project-diff/ - title: Project Diff - - - title: Merge to Production - url: /getting-started/branches/merge-to-production/ - - url: /kai/ title: Kai - AI Assistant items: @@ -343,7 +308,7 @@ items: title: FTP - url: /components/extractors/storage/google-drive/ - title: Google Drive + title: Google Sheets - url: /components/extractors/storage/http/ title: HTTP @@ -636,6 +601,27 @@ items: - url: /components/branches/merge-requests/ title: Merge Requests + - url: /components/branches/tutorial/ + title: Branches Tutorial + items: + - url: /components/branches/tutorial/prepare-tables/ + title: Prepare Tables + + - url: /components/branches/tutorial/prepare-files/ + title: Prepare Files + + - url: /components/branches/tutorial/tables-in-branch/ + title: Tables in Branch + + - url: /components/branches/tutorial/files-in-branch/ + title: Files in Branch + + - url: /components/branches/tutorial/project-diff/ + title: Project Diff + + - url: /components/branches/tutorial/merge-to-production/ + title: Merge to Production + - url: /components/ip-addresses/ title: IP Addresses @@ -798,6 +784,12 @@ items: - url: /workspace/snowflake-workspaces-access-changes/ title: Snowflake Workspaces Access Changes + - url: /workspace/create/ + title: Create a Workspace + + - url: /workspace/ad-hoc-analysis/ + title: Ad-Hoc Data Analysis + - url: /workspace/sql-editor/ title: SQL Editor @@ -877,6 +869,9 @@ items: - url: /ai/mcp-server/ title: MCP Server + - url: /ai/semantic-layer/ + title: Semantic Layer + - url: /extend/ title: Extending Keboola items: diff --git a/public/ai/semantic-layer/semantic-layer-metric.png b/public/ai/semantic-layer/semantic-layer-metric.png new file mode 100644 index 000000000..203ea4602 Binary files /dev/null and b/public/ai/semantic-layer/semantic-layer-metric.png differ diff --git a/public/ai/semantic-layer/semantic-layer-model.png b/public/ai/semantic-layer/semantic-layer-model.png new file mode 100644 index 000000000..77b96339e Binary files /dev/null and b/public/ai/semantic-layer/semantic-layer-model.png differ diff --git a/public/ai/semantic-layer/semantic-layer-models.png b/public/ai/semantic-layer/semantic-layer-models.png new file mode 100644 index 000000000..9d98918bd Binary files /dev/null and b/public/ai/semantic-layer/semantic-layer-models.png differ diff --git a/public/cli/token-create.png b/public/cli/token-create.png deleted file mode 100644 index 15cc4a273..000000000 Binary files a/public/cli/token-create.png and /dev/null differ diff --git a/public/getting-started/branches/bitcoin_price.csv b/public/components/branches/tutorial/bitcoin_price.csv similarity index 100% rename from public/getting-started/branches/bitcoin_price.csv rename to public/components/branches/tutorial/bitcoin_price.csv diff --git a/public/getting-started/branches/bitcoin_transactions.csv b/public/components/branches/tutorial/bitcoin_transactions.csv similarity index 100% rename from public/getting-started/branches/bitcoin_transactions.csv rename to public/components/branches/tutorial/bitcoin_transactions.csv diff --git a/public/getting-started/branches/figures/01-new-transformation.png b/public/components/branches/tutorial/figures/01-new-transformation.png similarity index 100% rename from public/getting-started/branches/figures/01-new-transformation.png rename to public/components/branches/tutorial/figures/01-new-transformation.png diff --git a/public/getting-started/branches/figures/05-output-mapping.png b/public/components/branches/tutorial/figures/05-output-mapping.png similarity index 100% rename from public/getting-started/branches/figures/05-output-mapping.png rename to public/components/branches/tutorial/figures/05-output-mapping.png diff --git a/public/getting-started/branches/figures/07-generated-file.png b/public/components/branches/tutorial/figures/07-generated-file.png similarity index 100% rename from public/getting-started/branches/figures/07-generated-file.png rename to public/components/branches/tutorial/figures/07-generated-file.png diff --git a/public/getting-started/branches/figures/08-create-dev-branch.png b/public/components/branches/tutorial/figures/08-create-dev-branch.png similarity index 100% rename from public/getting-started/branches/figures/08-create-dev-branch.png rename to public/components/branches/tutorial/figures/08-create-dev-branch.png diff --git a/public/getting-started/branches/figures/09-name-dev-branch.png b/public/components/branches/tutorial/figures/09-name-dev-branch.png similarity index 100% rename from public/getting-started/branches/figures/09-name-dev-branch.png rename to public/components/branches/tutorial/figures/09-name-dev-branch.png diff --git a/public/getting-started/branches/figures/10-dev-branch-created.png b/public/components/branches/tutorial/figures/10-dev-branch-created.png similarity index 100% rename from public/getting-started/branches/figures/10-dev-branch-created.png rename to public/components/branches/tutorial/figures/10-dev-branch-created.png diff --git a/public/getting-started/branches/figures/12-dev-branch-output.png b/public/components/branches/tutorial/figures/12-dev-branch-output.png similarity index 100% rename from public/getting-started/branches/figures/12-dev-branch-output.png rename to public/components/branches/tutorial/figures/12-dev-branch-output.png diff --git a/public/getting-started/branches/figures/14-jobs-log.png b/public/components/branches/tutorial/figures/14-jobs-log.png similarity index 100% rename from public/getting-started/branches/figures/14-jobs-log.png rename to public/components/branches/tutorial/figures/14-jobs-log.png diff --git a/public/getting-started/branches/figures/20-check-block1.png b/public/components/branches/tutorial/figures/20-check-block1.png similarity index 100% rename from public/getting-started/branches/figures/20-check-block1.png rename to public/components/branches/tutorial/figures/20-check-block1.png diff --git a/public/getting-started/branches/figures/21-storage-files-prod.png b/public/components/branches/tutorial/figures/21-storage-files-prod.png similarity index 100% rename from public/getting-started/branches/figures/21-storage-files-prod.png rename to public/components/branches/tutorial/figures/21-storage-files-prod.png diff --git a/public/getting-started/branches/figures/branch-deleted-storage.png b/public/components/branches/tutorial/figures/branch-deleted-storage.png similarity index 100% rename from public/getting-started/branches/figures/branch-deleted-storage.png rename to public/components/branches/tutorial/figures/branch-deleted-storage.png diff --git a/public/getting-started/branches/figures/branch-deleted.png b/public/components/branches/tutorial/figures/branch-deleted.png similarity index 100% rename from public/getting-started/branches/figures/branch-deleted.png rename to public/components/branches/tutorial/figures/branch-deleted.png diff --git a/public/getting-started/branches/figures/dev-branch-storage.png b/public/components/branches/tutorial/figures/dev-branch-storage.png similarity index 100% rename from public/getting-started/branches/figures/dev-branch-storage.png rename to public/components/branches/tutorial/figures/dev-branch-storage.png diff --git a/public/getting-started/branches/figures/diff-config-show-all.png b/public/components/branches/tutorial/figures/diff-config-show-all.png similarity index 100% rename from public/getting-started/branches/figures/diff-config-show-all.png rename to public/components/branches/tutorial/figures/diff-config-show-all.png diff --git a/public/getting-started/branches/figures/diff-config-show-changed.png b/public/components/branches/tutorial/figures/diff-config-show-changed.png similarity index 100% rename from public/getting-started/branches/figures/diff-config-show-changed.png rename to public/components/branches/tutorial/figures/diff-config-show-changed.png diff --git a/public/getting-started/branches/figures/extractor-output.png b/public/components/branches/tutorial/figures/extractor-output.png similarity index 100% rename from public/getting-started/branches/figures/extractor-output.png rename to public/components/branches/tutorial/figures/extractor-output.png diff --git a/public/getting-started/branches/figures/extractor-transactions-2.png b/public/components/branches/tutorial/figures/extractor-transactions-2.png similarity index 100% rename from public/getting-started/branches/figures/extractor-transactions-2.png rename to public/components/branches/tutorial/figures/extractor-transactions-2.png diff --git a/public/getting-started/branches/figures/extractor-transactions.png b/public/components/branches/tutorial/figures/extractor-transactions.png similarity index 100% rename from public/getting-started/branches/figures/extractor-transactions.png rename to public/components/branches/tutorial/figures/extractor-transactions.png diff --git a/public/getting-started/branches/figures/http-ex-prod-row.png b/public/components/branches/tutorial/figures/http-ex-prod-row.png similarity index 100% rename from public/getting-started/branches/figures/http-ex-prod-row.png rename to public/components/branches/tutorial/figures/http-ex-prod-row.png diff --git a/public/getting-started/branches/figures/http-ex-prod-set-up.png b/public/components/branches/tutorial/figures/http-ex-prod-set-up.png similarity index 100% rename from public/getting-started/branches/figures/http-ex-prod-set-up.png rename to public/components/branches/tutorial/figures/http-ex-prod-set-up.png diff --git a/public/getting-started/branches/figures/input-mapping-from-branch.png b/public/components/branches/tutorial/figures/input-mapping-from-branch.png similarity index 100% rename from public/getting-started/branches/figures/input-mapping-from-branch.png rename to public/components/branches/tutorial/figures/input-mapping-from-branch.png diff --git a/public/getting-started/branches/figures/mapping-in-branch-2.png b/public/components/branches/tutorial/figures/mapping-in-branch-2.png similarity index 100% rename from public/getting-started/branches/figures/mapping-in-branch-2.png rename to public/components/branches/tutorial/figures/mapping-in-branch-2.png diff --git a/public/getting-started/branches/figures/mapping-in-branch.png b/public/components/branches/tutorial/figures/mapping-in-branch.png similarity index 100% rename from public/getting-started/branches/figures/mapping-in-branch.png rename to public/components/branches/tutorial/figures/mapping-in-branch.png diff --git a/public/getting-started/branches/figures/merge-python-checkbox.png b/public/components/branches/tutorial/figures/merge-python-checkbox.png similarity index 100% rename from public/getting-started/branches/figures/merge-python-checkbox.png rename to public/components/branches/tutorial/figures/merge-python-checkbox.png diff --git a/public/getting-started/branches/figures/merge-python-dialog.png b/public/components/branches/tutorial/figures/merge-python-dialog.png similarity index 100% rename from public/getting-started/branches/figures/merge-python-dialog.png rename to public/components/branches/tutorial/figures/merge-python-dialog.png diff --git a/public/getting-started/branches/figures/merge-python-in-prod-2.png b/public/components/branches/tutorial/figures/merge-python-in-prod-2.png similarity index 100% rename from public/getting-started/branches/figures/merge-python-in-prod-2.png rename to public/components/branches/tutorial/figures/merge-python-in-prod-2.png diff --git a/public/getting-started/branches/figures/merge-python-in-prod.png b/public/components/branches/tutorial/figures/merge-python-in-prod.png similarity index 100% rename from public/getting-started/branches/figures/merge-python-in-prod.png rename to public/components/branches/tutorial/figures/merge-python-in-prod.png diff --git a/public/getting-started/branches/figures/merge-python-prod-storage.png b/public/components/branches/tutorial/figures/merge-python-prod-storage.png similarity index 100% rename from public/getting-started/branches/figures/merge-python-prod-storage.png rename to public/components/branches/tutorial/figures/merge-python-prod-storage.png diff --git a/public/getting-started/branches/figures/merge-snflk-dialog.png b/public/components/branches/tutorial/figures/merge-snflk-dialog.png similarity index 100% rename from public/getting-started/branches/figures/merge-snflk-dialog.png rename to public/components/branches/tutorial/figures/merge-snflk-dialog.png diff --git a/public/getting-started/branches/figures/new-snflk.png b/public/components/branches/tutorial/figures/new-snflk.png similarity index 100% rename from public/getting-started/branches/figures/new-snflk.png rename to public/components/branches/tutorial/figures/new-snflk.png diff --git a/public/getting-started/branches/figures/output-mapping-branch-transformation.png b/public/components/branches/tutorial/figures/output-mapping-branch-transformation.png similarity index 100% rename from public/getting-started/branches/figures/output-mapping-branch-transformation.png rename to public/components/branches/tutorial/figures/output-mapping-branch-transformation.png diff --git a/public/getting-started/branches/figures/partially-merged-branch.png b/public/components/branches/tutorial/figures/partially-merged-branch.png similarity index 100% rename from public/getting-started/branches/figures/partially-merged-branch.png rename to public/components/branches/tutorial/figures/partially-merged-branch.png diff --git a/public/getting-started/branches/figures/project-diff.png b/public/components/branches/tutorial/figures/project-diff.png similarity index 100% rename from public/getting-started/branches/figures/project-diff.png rename to public/components/branches/tutorial/figures/project-diff.png diff --git a/public/getting-started/branches/figures/python-branch-change-code.png b/public/components/branches/tutorial/figures/python-branch-change-code.png similarity index 100% rename from public/getting-started/branches/figures/python-branch-change-code.png rename to public/components/branches/tutorial/figures/python-branch-change-code.png diff --git a/public/getting-started/branches/figures/python-branch-overview.png b/public/components/branches/tutorial/figures/python-branch-overview.png similarity index 100% rename from public/getting-started/branches/figures/python-branch-overview.png rename to public/components/branches/tutorial/figures/python-branch-overview.png diff --git a/public/getting-started/branches/figures/python-new-codeblock.png b/public/components/branches/tutorial/figures/python-new-codeblock.png similarity index 100% rename from public/getting-started/branches/figures/python-new-codeblock.png rename to public/components/branches/tutorial/figures/python-new-codeblock.png diff --git a/public/getting-started/branches/figures/python-prod-overview.png b/public/components/branches/tutorial/figures/python-prod-overview.png similarity index 100% rename from public/getting-started/branches/figures/python-prod-overview.png rename to public/components/branches/tutorial/figures/python-prod-overview.png diff --git a/public/getting-started/branches/figures/show-project-diff.png b/public/components/branches/tutorial/figures/show-project-diff.png similarity index 100% rename from public/getting-started/branches/figures/show-project-diff.png rename to public/components/branches/tutorial/figures/show-project-diff.png diff --git a/public/getting-started/branches/figures/snflk-in-branch.png b/public/components/branches/tutorial/figures/snflk-in-branch.png similarity index 100% rename from public/getting-started/branches/figures/snflk-in-branch.png rename to public/components/branches/tutorial/figures/snflk-in-branch.png diff --git a/public/getting-started/branches/figures/snflk-new-table.png b/public/components/branches/tutorial/figures/snflk-new-table.png similarity index 100% rename from public/getting-started/branches/figures/snflk-new-table.png rename to public/components/branches/tutorial/figures/snflk-new-table.png diff --git a/public/getting-started/branches/figures/snflk-prod-code1.png b/public/components/branches/tutorial/figures/snflk-prod-code1.png similarity index 100% rename from public/getting-started/branches/figures/snflk-prod-code1.png rename to public/components/branches/tutorial/figures/snflk-prod-code1.png diff --git a/public/getting-started/branches/figures/snflk-prod-im.png b/public/components/branches/tutorial/figures/snflk-prod-im.png similarity index 100% rename from public/getting-started/branches/figures/snflk-prod-im.png rename to public/components/branches/tutorial/figures/snflk-prod-im.png diff --git a/public/getting-started/branches/figures/snflk-prod-om.png b/public/components/branches/tutorial/figures/snflk-prod-om.png similarity index 100% rename from public/getting-started/branches/figures/snflk-prod-om.png rename to public/components/branches/tutorial/figures/snflk-prod-om.png diff --git a/public/getting-started/branches/figures/storage-dev-buckets-2.png b/public/components/branches/tutorial/figures/storage-dev-buckets-2.png similarity index 100% rename from public/getting-started/branches/figures/storage-dev-buckets-2.png rename to public/components/branches/tutorial/figures/storage-dev-buckets-2.png diff --git a/public/getting-started/branches/figures/storage-dev-buckets.png b/public/components/branches/tutorial/figures/storage-dev-buckets.png similarity index 100% rename from public/getting-started/branches/figures/storage-dev-buckets.png rename to public/components/branches/tutorial/figures/storage-dev-buckets.png diff --git a/public/getting-started/branches/figures/transformation-branch-add-code.png b/public/components/branches/tutorial/figures/transformation-branch-add-code.png similarity index 100% rename from public/getting-started/branches/figures/transformation-branch-add-code.png rename to public/components/branches/tutorial/figures/transformation-branch-add-code.png diff --git a/public/getting-started/branches/figures/transformation-branch-added-code.png b/public/components/branches/tutorial/figures/transformation-branch-added-code.png similarity index 100% rename from public/getting-started/branches/figures/transformation-branch-added-code.png rename to public/components/branches/tutorial/figures/transformation-branch-added-code.png diff --git a/public/getting-started/branches/figures/transformation-branch-change-top-5.png b/public/components/branches/tutorial/figures/transformation-branch-change-top-5.png similarity index 100% rename from public/getting-started/branches/figures/transformation-branch-change-top-5.png rename to public/components/branches/tutorial/figures/transformation-branch-change-top-5.png diff --git a/public/getting-started/branches/figures/transformation-branch-input-mapping-missing.png b/public/components/branches/tutorial/figures/transformation-branch-input-mapping-missing.png similarity index 100% rename from public/getting-started/branches/figures/transformation-branch-input-mapping-missing.png rename to public/components/branches/tutorial/figures/transformation-branch-input-mapping-missing.png diff --git a/public/getting-started/branches/figures/transformation-branch-input-mapping.png b/public/components/branches/tutorial/figures/transformation-branch-input-mapping.png similarity index 100% rename from public/getting-started/branches/figures/transformation-branch-input-mapping.png rename to public/components/branches/tutorial/figures/transformation-branch-input-mapping.png diff --git a/public/getting-started/branches/figures/transformation-branch-output.png b/public/components/branches/tutorial/figures/transformation-branch-output.png similarity index 100% rename from public/getting-started/branches/figures/transformation-branch-output.png rename to public/components/branches/tutorial/figures/transformation-branch-output.png diff --git a/public/getting-started/branches/figures/transformation-branch-overview.png b/public/components/branches/tutorial/figures/transformation-branch-overview.png similarity index 100% rename from public/getting-started/branches/figures/transformation-branch-overview.png rename to public/components/branches/tutorial/figures/transformation-branch-overview.png diff --git a/public/getting-started/branches/figures/transformation-prod-set-up.png b/public/components/branches/tutorial/figures/transformation-prod-set-up.png similarity index 100% rename from public/getting-started/branches/figures/transformation-prod-set-up.png rename to public/components/branches/tutorial/figures/transformation-prod-set-up.png diff --git a/public/getting-started/load/access-to-spreadsheets.png b/public/components/extractors/storage/google-drive/access-to-spreadsheets.png similarity index 100% rename from public/getting-started/load/access-to-spreadsheets.png rename to public/components/extractors/storage/google-drive/access-to-spreadsheets.png diff --git a/public/getting-started/load/allow.png b/public/components/extractors/storage/google-drive/allow.png similarity index 100% rename from public/getting-started/load/allow.png rename to public/components/extractors/storage/google-drive/allow.png diff --git a/public/getting-started/load/find-spreadsheet.png b/public/components/extractors/storage/google-drive/find-spreadsheet.png similarity index 100% rename from public/getting-started/load/find-spreadsheet.png rename to public/components/extractors/storage/google-drive/find-spreadsheet.png diff --git a/public/getting-started/load/google-sheets-create.png b/public/components/extractors/storage/google-drive/google-sheets-create.png similarity index 100% rename from public/getting-started/load/google-sheets-create.png rename to public/components/extractors/storage/google-drive/google-sheets-create.png diff --git a/public/getting-started/load/google-sheets-spreadsheet.png b/public/components/extractors/storage/google-drive/google-sheets-spreadsheet.png similarity index 100% rename from public/getting-started/load/google-sheets-spreadsheet.png rename to public/components/extractors/storage/google-drive/google-sheets-spreadsheet.png diff --git a/public/getting-started/load/save-and-run.png b/public/components/extractors/storage/google-drive/save-and-run.png similarity index 100% rename from public/getting-started/load/save-and-run.png rename to public/components/extractors/storage/google-drive/save-and-run.png diff --git a/public/getting-started/load/select-files.png b/public/components/extractors/storage/google-drive/select-files.png similarity index 100% rename from public/getting-started/load/select-files.png rename to public/components/extractors/storage/google-drive/select-files.png diff --git a/public/getting-started/load/sign-in-with-google.png b/public/components/extractors/storage/google-drive/sign-in-with-google.png similarity index 100% rename from public/getting-started/load/sign-in-with-google.png rename to public/components/extractors/storage/google-drive/sign-in-with-google.png diff --git a/public/getting-started/load/source-intro-0.png b/public/components/extractors/storage/google-drive/source-intro-0.png similarity index 100% rename from public/getting-started/load/source-intro-0.png rename to public/components/extractors/storage/google-drive/source-intro-0.png diff --git a/public/getting-started/load/source-intro.png b/public/components/extractors/storage/google-drive/source-intro.png similarity index 100% rename from public/getting-started/load/source-intro.png rename to public/components/extractors/storage/google-drive/source-intro.png diff --git a/public/getting-started/load/storage.png b/public/components/extractors/storage/google-drive/storage.png similarity index 100% rename from public/getting-started/load/storage.png rename to public/components/extractors/storage/google-drive/storage.png diff --git a/public/data-apps/authentication/authentication.png b/public/data-apps/authentication/authentication.png deleted file mode 100644 index 18c79be4e..000000000 Binary files a/public/data-apps/authentication/authentication.png and /dev/null differ diff --git a/public/data-apps/authentication/select-oidc-provider.png b/public/data-apps/authentication/select-oidc-provider.png deleted file mode 100644 index 3164f5ef9..000000000 Binary files a/public/data-apps/authentication/select-oidc-provider.png and /dev/null differ diff --git a/public/getting-started/load/db-picture1.png b/public/getting-started/load/db-picture1.png deleted file mode 100644 index 533ef1955..000000000 Binary files a/public/getting-started/load/db-picture1.png and /dev/null differ diff --git a/public/getting-started/load/db-picture2.png b/public/getting-started/load/db-picture2.png deleted file mode 100644 index 8d7622710..000000000 Binary files a/public/getting-started/load/db-picture2.png and /dev/null differ diff --git a/public/getting-started/load/db-picture3.png b/public/getting-started/load/db-picture3.png deleted file mode 100644 index 2380789b3..000000000 Binary files a/public/getting-started/load/db-picture3.png and /dev/null differ diff --git a/public/getting-started/load/db-picture4.png b/public/getting-started/load/db-picture4.png deleted file mode 100644 index 64b0e89fb..000000000 Binary files a/public/getting-started/load/db-picture4.png and /dev/null differ diff --git a/public/getting-started/load/db-picture5.png b/public/getting-started/load/db-picture5.png deleted file mode 100644 index d613cbe8a..000000000 Binary files a/public/getting-started/load/db-picture5.png and /dev/null differ diff --git a/public/getting-started/load/db-picture6.png b/public/getting-started/load/db-picture6.png deleted file mode 100644 index ad23a520f..000000000 Binary files a/public/getting-started/load/db-picture6.png and /dev/null differ diff --git a/public/getting-started/load/db-picture7.png b/public/getting-started/load/db-picture7.png deleted file mode 100644 index 00ae84358..000000000 Binary files a/public/getting-started/load/db-picture7.png and /dev/null differ diff --git a/public/getting-started/load/db-picture8.png b/public/getting-started/load/db-picture8.png deleted file mode 100644 index ac27dc44d..000000000 Binary files a/public/getting-started/load/db-picture8.png and /dev/null differ diff --git a/public/getting-started/transform/workspaces1.png b/public/getting-started/transform/workspaces1.png deleted file mode 100644 index f26d1d943..000000000 Binary files a/public/getting-started/transform/workspaces1.png and /dev/null differ diff --git a/public/getting-started/transform/workspaces2.png b/public/getting-started/transform/workspaces2.png deleted file mode 100644 index 31d80a2e1..000000000 Binary files a/public/getting-started/transform/workspaces2.png and /dev/null differ diff --git a/public/getting-started/transform/workspaces3.png b/public/getting-started/transform/workspaces3.png deleted file mode 100644 index a1bbab478..000000000 Binary files a/public/getting-started/transform/workspaces3.png and /dev/null differ diff --git a/public/getting-started/transform/workspaces4.png b/public/getting-started/transform/workspaces4.png deleted file mode 100644 index 27e389deb..000000000 Binary files a/public/getting-started/transform/workspaces4.png and /dev/null differ diff --git a/public/getting-started/transform/workspaces5.png b/public/getting-started/transform/workspaces5.png deleted file mode 100644 index bdbc58b64..000000000 Binary files a/public/getting-started/transform/workspaces5.png and /dev/null differ diff --git a/public/getting-started/transform/workspaces6.png b/public/getting-started/transform/workspaces6.png deleted file mode 100644 index 74647dbb5..000000000 Binary files a/public/getting-started/transform/workspaces6.png and /dev/null differ diff --git a/public/getting-started/transform/workspaces7.png b/public/getting-started/transform/workspaces7.png deleted file mode 100644 index 51d2af795..000000000 Binary files a/public/getting-started/transform/workspaces7.png and /dev/null differ diff --git a/public/kai/kai-action-approval.png b/public/kai/kai-action-approval.png new file mode 100644 index 000000000..3a145e2e7 Binary files /dev/null and b/public/kai/kai-action-approval.png differ diff --git a/public/kai/kai-clarifying-question.png b/public/kai/kai-clarifying-question.png new file mode 100644 index 000000000..5571b6a2e Binary files /dev/null and b/public/kai/kai-clarifying-question.png differ diff --git a/public/kai/kai-close-chat.png b/public/kai/kai-close-chat.png new file mode 100644 index 000000000..831402d87 Binary files /dev/null and b/public/kai/kai-close-chat.png differ diff --git a/public/kai/kai-expand-chat.png b/public/kai/kai-expand-chat.png new file mode 100644 index 000000000..205fb9572 Binary files /dev/null and b/public/kai/kai-expand-chat.png differ diff --git a/public/kai/kai-follow-mode.png b/public/kai/kai-follow-mode.png new file mode 100644 index 000000000..f1b16820b Binary files /dev/null and b/public/kai/kai-follow-mode.png differ diff --git a/public/kai/kai-new-chat.png b/public/kai/kai-new-chat.png new file mode 100644 index 000000000..ef72e9d78 Binary files /dev/null and b/public/kai/kai-new-chat.png differ diff --git a/public/kai/kai-open.png b/public/kai/kai-open.png new file mode 100644 index 000000000..68d7f45cd Binary files /dev/null and b/public/kai/kai-open.png differ diff --git a/public/kai/kai-perm-always-allow.png b/public/kai/kai-perm-always-allow.png new file mode 100644 index 000000000..447faee07 Binary files /dev/null and b/public/kai/kai-perm-always-allow.png differ diff --git a/public/kai/kai-perm-always-ask.png b/public/kai/kai-perm-always-ask.png new file mode 100644 index 000000000..63dd5d639 Binary files /dev/null and b/public/kai/kai-perm-always-ask.png differ diff --git a/public/kai/kai-perm-block.png b/public/kai/kai-perm-block.png new file mode 100644 index 000000000..9811545be Binary files /dev/null and b/public/kai/kai-perm-block.png differ diff --git a/public/kai/kai-plan-mode.png b/public/kai/kai-plan-mode.png new file mode 100644 index 000000000..35390281a Binary files /dev/null and b/public/kai/kai-plan-mode.png differ diff --git a/public/kai/kai-report-bug.png b/public/kai/kai-report-bug.png new file mode 100644 index 000000000..8f77ab099 Binary files /dev/null and b/public/kai/kai-report-bug.png differ diff --git a/public/kai/kai-settings-context-files.png b/public/kai/kai-settings-context-files.png new file mode 100644 index 000000000..118a7cf81 Binary files /dev/null and b/public/kai/kai-settings-context-files.png differ diff --git a/public/kai/kai-settings-gear.png b/public/kai/kai-settings-gear.png new file mode 100644 index 000000000..fd0891c98 Binary files /dev/null and b/public/kai/kai-settings-gear.png differ diff --git a/public/kai/kai-settings-project-instructions.png b/public/kai/kai-settings-project-instructions.png new file mode 100644 index 000000000..ec698c4fd Binary files /dev/null and b/public/kai/kai-settings-project-instructions.png differ diff --git a/public/kai/kai-settings-skill-files.png b/public/kai/kai-settings-skill-files.png new file mode 100644 index 000000000..be2d8ca7a Binary files /dev/null and b/public/kai/kai-settings-skill-files.png differ diff --git a/public/kai/kai-settings-user-instructions.png b/public/kai/kai-settings-user-instructions.png new file mode 100644 index 000000000..a1fe4c1b2 Binary files /dev/null and b/public/kai/kai-settings-user-instructions.png differ diff --git a/public/kai/kai-skill-slash-menu.png b/public/kai/kai-skill-slash-menu.png new file mode 100644 index 000000000..e99211846 Binary files /dev/null and b/public/kai/kai-skill-slash-menu.png differ diff --git a/public/kai/kai-upload-file.png b/public/kai/kai-upload-file.png new file mode 100644 index 000000000..1ece54652 Binary files /dev/null and b/public/kai/kai-upload-file.png differ diff --git a/public/kai/kai-welcome.png b/public/kai/kai-welcome.png deleted file mode 100644 index 6370932dd..000000000 Binary files a/public/kai/kai-welcome.png and /dev/null differ diff --git a/public/getting-started/ad-hoc/cloud-platform-service-account-1.png b/public/workspace/ad-hoc-analysis/cloud-platform-service-account-1.png similarity index 100% rename from public/getting-started/ad-hoc/cloud-platform-service-account-1.png rename to public/workspace/ad-hoc-analysis/cloud-platform-service-account-1.png diff --git a/public/getting-started/ad-hoc/cloud-platform-service-account-3.png b/public/workspace/ad-hoc-analysis/cloud-platform-service-account-3.png similarity index 100% rename from public/getting-started/ad-hoc/cloud-platform-service-account-3.png rename to public/workspace/ad-hoc-analysis/cloud-platform-service-account-3.png diff --git a/public/getting-started/ad-hoc/cloud-platform-service-account-4.png b/public/workspace/ad-hoc-analysis/cloud-platform-service-account-4.png similarity index 100% rename from public/getting-started/ad-hoc/cloud-platform-service-account-4.png rename to public/workspace/ad-hoc-analysis/cloud-platform-service-account-4.png diff --git a/public/getting-started/ad-hoc/cloud-platform-service-account-5.png b/public/workspace/ad-hoc-analysis/cloud-platform-service-account-5.png similarity index 100% rename from public/getting-started/ad-hoc/cloud-platform-service-account-5.png rename to public/workspace/ad-hoc-analysis/cloud-platform-service-account-5.png diff --git a/public/getting-started/ad-hoc/cloud-platform-storage-1.png b/public/workspace/ad-hoc-analysis/cloud-platform-storage-1.png similarity index 100% rename from public/getting-started/ad-hoc/cloud-platform-storage-1.png rename to public/workspace/ad-hoc-analysis/cloud-platform-storage-1.png diff --git a/public/getting-started/ad-hoc/cloud-platform-storage-3.png b/public/workspace/ad-hoc-analysis/cloud-platform-storage-3.png similarity index 100% rename from public/getting-started/ad-hoc/cloud-platform-storage-3.png rename to public/workspace/ad-hoc-analysis/cloud-platform-storage-3.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-1.png b/public/workspace/ad-hoc-analysis/ex-bigquery-1.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-1.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-1.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-10.png b/public/workspace/ad-hoc-analysis/ex-bigquery-10.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-10.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-10.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-11.png b/public/workspace/ad-hoc-analysis/ex-bigquery-11.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-11.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-11.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-2.png b/public/workspace/ad-hoc-analysis/ex-bigquery-2.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-2.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-2.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-3.png b/public/workspace/ad-hoc-analysis/ex-bigquery-3.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-3.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-3.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-4.png b/public/workspace/ad-hoc-analysis/ex-bigquery-4.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-4.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-4.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-5.png b/public/workspace/ad-hoc-analysis/ex-bigquery-5.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-5.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-5.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-6.png b/public/workspace/ad-hoc-analysis/ex-bigquery-6.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-6.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-6.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-7.png b/public/workspace/ad-hoc-analysis/ex-bigquery-7.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-7.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-7.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-8.png b/public/workspace/ad-hoc-analysis/ex-bigquery-8.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-8.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-8.png diff --git a/public/getting-started/ad-hoc/ex-bigquery-9.png b/public/workspace/ad-hoc-analysis/ex-bigquery-9.png similarity index 100% rename from public/getting-started/ad-hoc/ex-bigquery-9.png rename to public/workspace/ad-hoc-analysis/ex-bigquery-9.png diff --git a/public/getting-started/ad-hoc/sandbox-1.png b/public/workspace/ad-hoc-analysis/sandbox-1.png similarity index 100% rename from public/getting-started/ad-hoc/sandbox-1.png rename to public/workspace/ad-hoc-analysis/sandbox-1.png diff --git a/public/getting-started/ad-hoc/sandbox-2.png b/public/workspace/ad-hoc-analysis/sandbox-2.png similarity index 100% rename from public/getting-started/ad-hoc/sandbox-2.png rename to public/workspace/ad-hoc-analysis/sandbox-2.png diff --git a/public/getting-started/ad-hoc/sandbox-3.png b/public/workspace/ad-hoc-analysis/sandbox-3.png similarity index 100% rename from public/getting-started/ad-hoc/sandbox-3.png rename to public/workspace/ad-hoc-analysis/sandbox-3.png diff --git a/public/getting-started/ad-hoc/transformation-1.png b/public/workspace/ad-hoc-analysis/transformation-1.png similarity index 100% rename from public/getting-started/ad-hoc/transformation-1.png rename to public/workspace/ad-hoc-analysis/transformation-1.png diff --git a/public/getting-started/ad-hoc/transformation-2.png b/public/workspace/ad-hoc-analysis/transformation-2.png similarity index 100% rename from public/getting-started/ad-hoc/transformation-2.png rename to public/workspace/ad-hoc-analysis/transformation-2.png diff --git a/public/getting-started/ad-hoc/transformation-3.png b/public/workspace/ad-hoc-analysis/transformation-3.png similarity index 100% rename from public/getting-started/ad-hoc/transformation-3.png rename to public/workspace/ad-hoc-analysis/transformation-3.png diff --git a/public/getting-started/ad-hoc/transformation-4.png b/public/workspace/ad-hoc-analysis/transformation-4.png similarity index 100% rename from public/getting-started/ad-hoc/transformation-4.png rename to public/workspace/ad-hoc-analysis/transformation-4.png diff --git a/public/workspace/create/workspaces1.png b/public/workspace/create/workspaces1.png new file mode 100644 index 000000000..15f615c13 Binary files /dev/null and b/public/workspace/create/workspaces1.png differ diff --git a/public/workspace/create/workspaces2.png b/public/workspace/create/workspaces2.png new file mode 100644 index 000000000..3a950b207 Binary files /dev/null and b/public/workspace/create/workspaces2.png differ diff --git a/public/workspace/create/workspaces3.png b/public/workspace/create/workspaces3.png new file mode 100644 index 000000000..7b7e2d4e6 Binary files /dev/null and b/public/workspace/create/workspaces3.png differ diff --git a/public/workspace/create/workspaces4.png b/public/workspace/create/workspaces4.png new file mode 100644 index 000000000..c5d58072c Binary files /dev/null and b/public/workspace/create/workspaces4.png differ diff --git a/public/workspace/create/workspaces5.png b/public/workspace/create/workspaces5.png new file mode 100644 index 000000000..2ebc24100 Binary files /dev/null and b/public/workspace/create/workspaces5.png differ diff --git a/public/workspace/create/workspaces6.png b/public/workspace/create/workspaces6.png new file mode 100644 index 000000000..7f11a95cb Binary files /dev/null and b/public/workspace/create/workspaces6.png differ diff --git a/public/workspace/create/workspaces7.png b/public/workspace/create/workspaces7.png new file mode 100644 index 000000000..666e9310b Binary files /dev/null and b/public/workspace/create/workspaces7.png differ diff --git a/src/content/docs/ai/ai-kit/index.md b/src/content/docs/ai/ai-kit/index.md index 145a16caf..71f4e6f42 100644 --- a/src/content/docs/ai/ai-kit/index.md +++ b/src/content/docs/ai/ai-kit/index.md @@ -7,9 +7,9 @@ slug: 'ai/ai-kit' AI Kit is a plugin marketplace for AI coding assistants that provides specialized agents, commands, and workflows for Keboola development. It helps developers build Keboola components, data apps, and maintain code quality using AI-powered tools. -AI Kit is designed for developers who use AI coding assistants like Claude Code to work with Keboola projects. It provides three specialized plugins that cover different aspects of Keboola development, from general code quality to building production-ready components and data applications. +AI Kit is designed for developers who use AI coding assistants like Claude Code to work with Keboola projects. It provides seven plugins that cover different aspects of Keboola development, from building production-ready components and data applications to driving your projects from the terminal and modelling your semantic layer. -The toolkit includes specialized AI agents that understand Keboola's architecture, best practices, and development patterns. These agents can help you create new components from scratch, implement configuration schemas, build Streamlit data apps, review code for security issues, and automate common development workflows. +The toolkit includes specialized AI agents that understand Keboola's architecture, best practices, and development patterns. These agents can help you create new components from scratch, implement configuration schemas, build data apps, review a project for SQL and security problems, and automate common development workflows. ## Installation @@ -22,36 +22,26 @@ To install AI Kit, run the following command in your AI coding assistant: After installation, enable the plugins you need: ```bash -/plugin install developer -/plugin install component-developer -/plugin install dataapp-developer +/plugin install component-developer@keboola-claude-kit +/plugin install dataapp-developer@keboola-claude-kit +/plugin install kbagent@keboola-claude-kit +/plugin install keboola-cli@keboola-claude-kit +/plugin install keboola-git@keboola-claude-kit +/plugin install powerbi-to-sl@keboola-claude-kit +/plugin install sl-toolkit@keboola-claude-kit ``` -## Available Plugins - -### Developer Plugin - -The Developer Plugin provides a comprehensive toolkit for code quality, security analysis, and workflow automation. It includes four specialized agents and a PR creation command. - -**Agents:** - -The **Code Reviewer** (`@code-reviewer`) is an expert code reviewer that checks for bugs, logic errors, security vulnerabilities, and project guidelines compliance. It uses confidence-based filtering to report only high-priority issues. +`keboola-claude-kit` is the marketplace name this repository publishes, and the one to install Keboola plugins from. -The **Security Agent** (`@code-security`) performs cross-language security analysis for Python, Go, PHP, JavaScript, and other languages. It integrates with automated security scanners and identifies CWE/CVE vulnerabilities with actionable fixes. + -The **Code Mess Detector** (`@code-mess-detector`) analyzes code written during rapid prototyping for common quality issues like inconsistent naming, missing error handling, code duplication, and dead code. It generates detailed reports for systematic cleanup. - -The **Code Mess Fixer** (`@code-mess-fixer`) systematically applies fixes based on Code Mess Detector reports, working through issues by priority and tracking progress. - -**Commands:** - -The `/create-pr` command analyzes your changes and creates a pull request with an AI-generated title and description, following conventional commit format. +## Available Plugins -**Integrations:** +### kbagent Plugin -The plugin includes Linear MCP integration for issue tracking and project management, and auto-installs team-wide permission settings for safe git operations. +The kbagent Plugin teaches an AI client the [kbagent CLI](/cli/), so the assistant can run jobs, read configurations and search across your Keboola projects. It adds a `/keboola` command backed by a `keboola-expert` subagent, a skill that understands plain-language asks like "set up kbagent for my Keboola project", and in Claude Code a `/kbagent:setup` command that installs and connects the CLI in one step. -[View Developer Plugin Documentation on GitHub](https://github.com/keboola/ai-kit/tree/main/plugins/developer) +The plugin's source lives in the CLI repo at [`plugins/kbagent`](https://github.com/keboola/cli/tree/main/plugins/kbagent); this marketplace publishes it from there through a `git-subdir` source. Claude Code, Claude Desktop, Cursor, VS Code and the ChatGPT app can all install it, each by its own route. See [kbagent with AI agents](/cli/for-agents/) for the steps per client. ### Component Developer Plugin @@ -81,6 +71,30 @@ The component should support incremental loads based on a timestamp field. [View Component Developer Plugin Documentation on GitHub](https://github.com/keboola/ai-kit/tree/main/plugins/component-developer) +### Keboola CLI Plugin + +The Keboola CLI Plugin is a project management and review toolkit built on the older `kbc` sync CLI. It ships a ten-agent review team that analyses a project for SQL quality, security, performance, financial logic, and template readiness. It is a separate tool from the kbagent plugin above. If you want the agent interface to your projects, install `kbagent`. + +[View Keboola CLI Plugin Documentation on GitHub](https://github.com/keboola/ai-kit/tree/main/plugins/keboola-cli) + +### Keboola Git Plugin + +The Keboola Git Plugin works with Keboola-managed Git (Forgejo) repositories for Python/JS data apps, which can host their source in Keboola rather than GitHub. It provisions repositories, mints push credentials, and copies source between GitHub and Keboola git through the kbagent CLI. It also carries the 15 MB push cap and the build-at-deploy workaround it forces. + +[View Keboola Git Plugin Documentation on GitHub](https://github.com/keboola/ai-kit/tree/main/plugins/keboola-git) + +### Semantic Layer Toolkit + +The `sl-toolkit` plugin inspects, validates, and builds semantic layer models through the metastore API. `/sl-show` lists a model's datasets, metrics, and relationships, `/sl-validate` checks it for phantom fields and dangling references, and `/sl-build` walks a greenfield model from schema discovery to push. Adding and editing model objects works conversationally, with no slash command. + +[View Semantic Layer Toolkit Documentation on GitHub](https://github.com/keboola/ai-kit/tree/main/plugins/sl-toolkit) + +### Power BI to Semantic Layer + +The `powerbi-to-sl` plugin migrates an existing Microsoft Power BI semantic model into a Keboola semantic layer model, translating tables, columns, measures, and relationships. It is the brownfield companion to `sl-toolkit`, which generates a new model instead. + +[View Power BI to Semantic Layer Documentation on GitHub](https://github.com/keboola/ai-kit/tree/main/plugins/powerbi-to-sl) + ### Data App Developer Plugin The Data App Developer Plugin is a specialized toolkit for building production-ready Streamlit data apps for Keboola deployment. It features a systematic validate, build, and verify workflow that ensures features work correctly the first time. @@ -108,7 +122,7 @@ The agent will automatically validate the schema, query distinct values, create ## Best Practices -When using AI Kit, start with the appropriate plugin for your task. Use the Developer Plugin for general code quality and security reviews, the Component Developer Plugin when building new Keboola components or adding features to existing ones, and the Data App Developer Plugin when creating or modifying Streamlit data apps. +When using AI Kit, start with the appropriate plugin for your task. Use the Component Developer Plugin when building new Keboola components or adding features to existing ones, the Data App Developer Plugin when creating or modifying data apps, the kbagent Plugin to drive your projects from the terminal, and the Semantic Layer Toolkit when modelling metrics. For component development, always follow the two-PR workflow strategy: create a base PR with the cookiecutter-generated structure, then a separate implementation PR with your custom logic. This prevents premature CI/CD triggers. diff --git a/src/content/docs/ai/index.md b/src/content/docs/ai/index.md index 8a7dc8954..57bff9406 100644 --- a/src/content/docs/ai/index.md +++ b/src/content/docs/ai/index.md @@ -27,6 +27,11 @@ A plugin marketplace for AI coding assistants that provides specialized agents, Model Context Protocol server integration for seamless communication between AI agents and your data infrastructure on Keboola. [Learn more about MCP Server →](/ai/mcp-server/) +### Semantic Layer + +Describe your data in business terms — datasets, metrics, relationships, glossary terms, and business rules — so AI assistants understand what your data means, not just how it is stored. +[Learn more about the Semantic Layer →](/ai/semantic-layer/) + ### Machine-Readable API Index Every Keboola stack publishes a machine-readable index of its APIs at `https://api./apis.json` (for example, [`https://api.keboola.com/apis.json`](https://api.keboola.com/apis.json)). The index lists each service with its base `apiUrl` and a link to its OpenAPI specification (`openApiSpecUrl`) — a convenient overview for agentic usage: AI agents, MCP servers, and other tooling that discovers Keboola's APIs programmatically. diff --git a/src/content/docs/ai/mcp-server/index.md b/src/content/docs/ai/mcp-server/index.md index 7c06b221d..e47b0f804 100644 --- a/src/content/docs/ai/mcp-server/index.md +++ b/src/content/docs/ai/mcp-server/index.md @@ -249,7 +249,7 @@ Don't worry about remembering command names — your AI client handles that. Jus - **Components & Transformations** – Create, edit, and launch them with natural language. - **Storage** – Browse, edit, and document buckets, tables, and columns. - **SQL** – Run and manage SQL queries. -- **Semantic layer** – Explore the project's semantic models and validate queries against them. +- **Semantic layer** – Explore the project's semantic models and validate queries against them. See [Semantic Layer](/ai/semantic-layer/). - **Jobs** – Start, monitor, and debug execution flows. - **Flows** – Create and manage flows (including conditional flows) that orchestrate your components. - **Data Apps** – Create, deploy, and manage Streamlit and Python/JS data apps. @@ -301,13 +301,13 @@ The following tools are classified as read-only (they do not modify data). The l |----------|-------| | Components | `get_configs`, `get_components`, `get_config_examples`, `run_sync_action` | | Flows | `get_flows`, `get_flow_examples`, `get_flow_schema` | -| Storage | `get_buckets`, `get_tables` | +| Storage | `get_buckets`, `get_shared_buckets`, `get_tables` | | SQL | `query_data` | | Semantic | `get_semantic_context`, `get_semantic_schema`, `search_semantic_context`, `validate_semantic_query` | | Data Apps | `get_data_apps` | | Jobs | `get_jobs` | | Search | `search`, `find_component_id` | -| Project | `get_project_info` | +| Project | `get_accessible_projects`, `get_project_info`, `set_project_scope` | | Documentation | `docs_query` | ### Examples @@ -436,7 +436,7 @@ The primary way to run the server locally without Docker is by using `uv` or `uv * `KBC_STORAGE_API_URL`: Your Keboola instance API URL (e.g., `https://connection.keboola.com` or `https://connection.YOUR_REGION.keboola.com`). * `KBC_BRANCH_ID` (optional): a development branch ID to scope operations to; defaults to the production branch. - Refer to the [Keboola Tokens](/management/project/tokens/) and [Keboola workspace manipulation](/getting-started/transform/workspace/) for detailed instructions on obtaining these values. + Refer to the [Keboola Tokens](/management/project/tokens/) and [Keboola workspace manipulation](/workspace/create/) for detailed instructions on obtaining these values. **1.1. Additional Setup for BigQuery Users** If your Keboola project uses BigQuery as its backend, you will also need to set up the `GOOGLE_APPLICATION_CREDENTIALS` environment variable. This variable should point to the JSON file containing your Google Cloud service account key that has the necessary permissions to access your BigQuery data. @@ -579,7 +579,7 @@ For detailed instructions and SDKs for building your own MCP client, refer to th ## Advanced Setup Options These methods are for developers or specific use cases (e.g., testing, contributing to the MCP server). -Prefer a terminal or want to give an agent sandboxed, multi-project control? See the [kbagent CLI](/cli/) — it can also call MCP tools via `kbagent tool`. For dev environments or contributing to the MCP Server, check out the [MCP GitHub repo](https://github.com/keboola/mcp-server). +Prefer a terminal or want to give an agent sandboxed, multi-project control? See the [kbagent CLI](/cli/). For dev environments or contributing to the MCP Server, check out the [MCP GitHub repo](https://github.com/keboola/mcp-server). ## Support and Feedback diff --git a/src/content/docs/ai/semantic-layer/index.md b/src/content/docs/ai/semantic-layer/index.md new file mode 100644 index 000000000..4c18ad6bc --- /dev/null +++ b/src/content/docs/ai/semantic-layer/index.md @@ -0,0 +1,212 @@ +--- +title: Semantic Layer +slug: 'ai/semantic-layer' +description: Describe your data in business terms — datasets, metrics, relationships, glossary terms, and business rules — so AI assistants understand what your data means. +--- + +:::caution[Beta] +The semantic layer is in beta. It runs on Keboola's multi-tenant stacks and is not offered on +single-tenant stacks. The **Semantic Layer** section in the UI is enabled per project separately +from the semantic layer itself, so your project can already hold a semantic model — one built by +Kai, for example — before the section appears. If your project has no **Semantic Layer** section, +contact our [support team](mailto:support@keboola.com). +::: + + + +Keboola's semantic layer lets you describe your project's data in business terms — datasets, metrics, relationships, glossary terms, and business rules. [Kai](/kai/), AI assistants connected to your project through the [MCP Server](/ai/mcp-server/), and the [Keboola CLI](/cli/) all read these definitions to understand what your data *means*, not just how it is stored. + +Instead of every AI conversation having to rediscover which table holds revenue, how orders join to customers, or which business rules a query must respect, you define these facts once. Every AI assistant working with your project then grounds its answers — and the SQL it generates — in the same shared definitions. + +## Why use a semantic layer? + +- **Consistent answers** – A metric such as "net revenue" is defined once, as a SQL expression, and every AI-generated query uses the same definition. +- **Business vocabulary** – Glossary terms teach the AI your company's language, so questions asked in business terms resolve to the right data. +- **Guardrails for AI-generated SQL** – Constraints capture business rules (for example, "profit must never exceed revenue"), and queries can be validated against them before they are executed. +- **Less schema exploration** – The AI spends less time inspecting raw tables and columns because the relevant context is already curated. + +## Core concepts + +A **semantic model** is a collection of semantic objects stored centrally in Keboola. Six semantic object types make up a model: + +| Object type | What it describes | +|---|---| +| `semantic-model` | The top-level container for a set of semantic definitions. It also records the SQL dialect used by the model's SQL expressions. | +| `semantic-dataset` | Maps a Keboola table (by table ID) to a business entity, including its fields and primary key. | +| `semantic-metric` | A named business calculation defined as a SQL expression over a dataset — for example, revenue, order count, or margin. | +| `semantic-relationship` | How two datasets join: the from/to datasets, the join type, and the join condition. | +| `semantic-glossary` | A business term and its definition — your company vocabulary. | +| `semantic-constraint` | A business rule with a severity (`error`, `warning`, or `info`) that queries can be checked against. | + +A project can contain multiple semantic models. Each object is a JSON document validated against a published JSON schema. + +There is also a seventh type, `semantic-reference-data` — a per-dimension member store holding the +full member list for a dimension, such as a chart of accounts. It is not part of a model's build, +export, or diff, and is managed with `kbagent semantic-layer reference-data`. + +## Building a semantic model + +Start here: a project with no semantic model has nothing for an AI assistant to ground on, and the +semantic MCP tools stay hidden until at least one model exists. There are four ways to build one. + +### With Kai + +The quickest way is to ask [Kai](/kai/), the assistant built into Keboola. Kai builds and maintains +semantic models from a chat inside your project, with nothing to install: + +> "Build a semantic model from the tables in the `out.c-sales` bucket." +> "Add a net profit margin metric to the sales model, and a rule that flags a margin above 100%." + +Kai reads your buckets and tables, asks which numbers the business actually tracks, drafts the +model, and validates it before writing anything. It then asks for your approval, and the approval +card names the model and how many objects of each type the call will write, so you see the whole +change before it happens. Editing, removing, and sharing objects in a model you already have work +the same way: describe the change in plain language, review it, approve it. See +[Action approval](/kai/getting-started/#action-approval). + +A model Kai creates is visible only in the project it was created in. Kai can also share it +read-only with named sibling projects, or with every project in your organization; widening a model +to the whole organization requires an organization admin. + + + +### In the Keboola UI + +Your project's **Semantic Layer** section lists the project's semantic models: + +![The Semantic Layer section in the Keboola UI, listing the project's semantic models](/ai/semantic-layer/semantic-layer-models.png) + +A model opens as one tab per object type — datasets, metrics, constraints, relationships, and +glossary terms: + +![A semantic model opened in the UI, with one tab per semantic object type](/ai/semantic-layer/semantic-layer-model.png) + +Objects open read-only, showing exactly the definition an AI assistant reads — a metric, for +example, shows its SQL expression and description — and are edited explicitly via **Edit**. A +**Metadata** tab tracks the object's revision, schema version, and branch: + +![A metric opened read-only in the UI, with its SQL expression, description, and an Edit button](/ai/semantic-layer/semantic-layer-metric.png) + + + +### With the CLI + +The [Keboola CLI](/cli/commands/) carries a `semantic-layer` command group covering the whole +lifecycle without an AI in the loop — `build` a model from a list of storage tables, `show`, +`export`, `diff`, `validate`, `promote` a model between projects, and add or edit individual +metrics, datasets, relationships, constraints, and glossary terms. It also reads the model the way +an assistant does, with `search-context` and `get-context`: + +```bash +kbagent semantic-layer --help +``` + +### With AI Kit plugins + +Two [AI Kit](/ai/ai-kit/) plugins cover the two most common starting points from an AI coding +assistant such as Claude Code: building a model from scratch, and migrating one you already have. + +#### Semantic Layer Toolkit (`sl-toolkit`) + +The Semantic Layer Toolkit lets you build, inspect, validate, and edit semantic models from your assistant. + +**Commands:** + +- `/sl-build` – A greenfield wizard that builds a new semantic model from your Keboola project: schema discovery → SQL analysis → generation → validation → push. +- `/sl-show` – Lists all datasets, metrics, relationships, constraints, and glossary terms in a model. +- `/sl-validate` – Checks a model for consistency issues such as references to non-existent fields or dangling relationships. + +**Conversational editing:** + +Adding, editing, and removing semantic objects doesn't need commands — just describe the change: + +> "Add a metric for net profit margin on the KPI dashboard table." +> "Rename the Revenue metric to Total Revenue." + +:::note +`/sl-build` and `/sl-validate --deep` read your project's schemas through the `kbagent` binary from +the [Keboola CLI](/cli/getting-started/). Without it on your `PATH` they still run, but `/sl-build` +asks you to describe your tables instead of reading them, and the deep checks are skipped. +`/sl-show`, plain `/sl-validate`, and conversational edits don't use it at all. +::: + +[View the Semantic Layer Toolkit on GitHub](https://github.com/keboola/ai-kit/tree/main/plugins/sl-toolkit) + +#### Power BI migration (`powerbi-to-sl`) + +If you already maintain a semantic model in Microsoft Power BI, the `powerbi-to-sl` plugin translates it into Keboola semantic layer objects: Power BI tables become semantic datasets, measures become semantic metrics (DAX expressions are preserved verbatim for review), and relationships become semantic relationships. The recommended input is a TMDL export produced by Microsoft's Power BI Modeling MCP server in read-only mode. + +The plugin flags anything that needs human attention — such as complex DAX or unmapped data types — in a warnings report. Pushing the result to your project is not automatic — hand it to `sl-toolkit`, or push it yourself. + +[View the Power BI migration plugin on GitHub](https://github.com/keboola/ai-kit/tree/main/plugins/powerbi-to-sl) + +#### Installing the plugins + +Both plugins are installed from the [AI Kit](/ai/ai-kit/) marketplace: + +```bash +/plugin marketplace add keboola/ai-kit +/plugin install sl-toolkit +/plugin install powerbi-to-sl +``` + +## Using the semantic layer via MCP + +Once your project contains at least one semantic model, four additional tools appear in the +[Keboola MCP Server](/ai/mcp-server/). All of them are read-only. + +:::note +The tools are hidden while a project has no semantic model — the server checks per request and +keeps them hidden if it cannot reach the semantic layer. If your assistant reports that the +semantic tools are unavailable, build a model first. +::: + +| Tool | What it does | +|---|---| +| `search_semantic_context` | Searches semantic models and objects using regex patterns matched against names, descriptions, and attributes. Used to discover which semantic objects are relevant to a question. | +| `get_semantic_context` | Loads semantic objects by type — all objects of a type in compact form, or specific objects by ID with full attributes. | +| `get_semantic_schema` | Returns the published schema information for a semantic object type. It currently reports the available schema versions rather than the schema document itself; `kbagent semantic-layer schema` resolves the default version and returns the full JSON Schema. | +| `validate_semantic_query` | Performs a best-effort semantic validation of a SQL query against one or more semantic models: it detects which datasets, metrics, and relationships the query uses and surfaces constraint violations — without executing the query. | + +You don't call these tools yourself. Ask questions in plain language ("What was our net revenue last quarter, by region?") and your AI assistant uses them to ground its answer: + +1. **Discover** – `search_semantic_context` finds the semantic objects related to your question, such as the "net revenue" metric and the datasets it is built on. +2. **Load** – `get_semantic_context` retrieves the full definitions of the relevant objects. +3. **Validate** – Before running any SQL, `validate_semantic_query` checks the query against the model and reports business-rule violations. +4. **Query** – The assistant executes the validated SQL with the standard `query_data` tool. + +:::note +Semantic query validation is heuristic — it matches the SQL text against semantic metadata rather than fully parsing the query. Treat it as a best-effort check, not a formal proof of correctness. +::: + +Because these four tools are read-only, they remain available when the MCP connection is restricted with the `X-Read-Only-Mode` header (see [Restricting Tool Access](/ai/mcp-server/#restricting-tool-access)). + +## Example prompts + +Once your project has a populated semantic model and your AI assistant is connected via MCP, try: + +- "What semantic models are defined in this project?" +- "What was our total revenue last month? Use the semantic layer definitions." +- "Which business rules apply to queries on the orders dataset?" +- "Validate this SQL against the sales semantic model before running it." + +## Support and feedback + +If you run into issues or have feedback during the beta, contact our [support team](mailto:support@keboola.com) — beta feedback directly shapes where the semantic layer goes next. diff --git a/src/content/docs/ai/semantic-layer/semantic-layer-metric.png b/src/content/docs/ai/semantic-layer/semantic-layer-metric.png new file mode 100644 index 000000000..203ea4602 Binary files /dev/null and b/src/content/docs/ai/semantic-layer/semantic-layer-metric.png differ diff --git a/src/content/docs/ai/semantic-layer/semantic-layer-model.png b/src/content/docs/ai/semantic-layer/semantic-layer-model.png new file mode 100644 index 000000000..77b96339e Binary files /dev/null and b/src/content/docs/ai/semantic-layer/semantic-layer-model.png differ diff --git a/src/content/docs/ai/semantic-layer/semantic-layer-models.png b/src/content/docs/ai/semantic-layer/semantic-layer-models.png new file mode 100644 index 000000000..9d98918bd Binary files /dev/null and b/src/content/docs/ai/semantic-layer/semantic-layer-models.png differ diff --git a/src/content/docs/cli/commands.md b/src/content/docs/cli/commands.md index 2d2d54e34..e0c1d40bc 100644 --- a/src/content/docs/cli/commands.md +++ b/src/content/docs/cli/commands.md @@ -69,7 +69,6 @@ $ kbagent --json project list - **`workspace`** — `create`, `list`, `detail`, `delete`, `password`, `load`, `query`, `from-transformation`, `gc`. - **`sync`** — `init`, `pull`, `status`, `diff`, `push`, `clone`, `branch-link/unlink/status`. - **`encrypt`** — `values` (one-way encrypt `#`-prefixed secrets). -- **`tool`** — call Keboola [MCP](/ai/mcp-server/) tools directly. - **`agent`** — **scheduled AI agents**: cron-driven agent tasks that run against your projects unattended ([recipe](/cli/workflows/#schedule-an-ai-agent)): `list`, `show`, `create`, `update`, `delete`, `run`, `runs`, `run-detail`, `run-events`, `test`, `prompt-improve`, `cron-preview`. - **`semantic-layer`**, **`dev-portal`**, **`http`**. diff --git a/src/content/docs/cli/for-agents.md b/src/content/docs/cli/for-agents.md index 4c2942c46..ab762166f 100644 --- a/src/content/docs/cli/for-agents.md +++ b/src/content/docs/cli/for-agents.md @@ -1,29 +1,129 @@ --- -title: kbagent for AI agents +title: kbagent with AI agents slug: 'cli/for-agents' sidebar: label: Use with AI Agents -description: 'Give an AI coding agent safe control of Keboola with kbagent — the Claude Code plugin and /keboola subagent, read-only sandboxing, the conversation ID, and the kbagent context reference.' +description: 'Give an AI coding agent safe control of Keboola with kbagent: the plugin and /keboola subagent, per-client setup for Claude Code, Claude Desktop, Cursor, VS Code and the ChatGPT app, read-only sandboxing, the conversation ID, and the kbagent context reference.' --- -[kbagent](/cli/) is built to be driven by AI coding agents (Claude Code, Cursor, Copilot), not just humans. It gives an agent a stable command surface, a machine-readable reference, and safety rails so it can operate Keboola without you handing over unrestricted access. +[kbagent](/cli/) is built to be driven by AI coding agents (Claude Code, Claude Desktop, Cursor, VS Code, the ChatGPT app) as well as humans. It gives an agent a stable command surface, a machine-readable reference, and safety rails so it can operate Keboola without you handing over unrestricted access. - + -## Claude Code plugin +## The kbagent plugin -:::tip[Add kbagent to Claude Code] -Run these two commands **inside Claude Code** to install the plugin from the CLI's own marketplace (use the copy button on the block): +The plugin teaches an AI client the CLI. It adds a **`/keboola`** slash command that spawns a `keboola-expert` subagent with fresh context and hard rules (fetch the current reference, dry-run first, prefer the CLI over raw REST/MCP, gate on version), plus a structured verification payload. It also ships a skill, so where the plugin is installed you can just ask for what you want in plain language ("set up kbagent for my Keboola project", "log me out of Keboola") instead of copying commands. -``` -/plugin marketplace add keboola/cli -/plugin install kbagent@keboola-agent-cli -``` +The plugin lives in the CLI repo at [`plugins/kbagent`](https://github.com/keboola/cli/tree/main/plugins/kbagent) and ships from Keboola's [AI Kit](/ai/ai-kit/) marketplace, which is named `keboola-claude-kit`. + +Five clients can install it: Claude Code, Claude Desktop, Cursor, VS Code and the ChatGPT app. Each has its own route, and all five are below. The plugin does not connect your project. That is a separate step, so you need both. + +In every client's plugin list, **`keboola-cli`** sits next to `kbagent`. It is a separate project-review toolkit built on the older `kbc` sync CLI. For the agent interface described here, pick `kbagent`. + +:::caution[Added the marketplace from keboola/cli before?] +Earlier versions of this page pointed at `keboola/cli`. Nothing moves you off it. Run `/plugin marketplace remove keboola-agent-cli` first, then add `keboola/ai-kit` and install the plugin again. `kbagent doctor` names the marketplace you are on. ::: -The plugin installs via Claude Code's marketplace, not as a downloaded file — so it's a command you run in the assistant, not a button. It adds a **`/keboola`** slash command that spawns a `keboola-expert` subagent with fresh context and hard rules (fetch the current reference, dry-run first, prefer the CLI over raw REST/MCP, gate on version), plus a structured verification payload. `kbagent doctor` tells you whether the plugin is installed. +### The `/kbagent:setup` shortcut + +Claude Code is the only client that can run **`/kbagent:setup`**. Pass it a stack URL. It installs the CLI if it is missing, then signs you in itself: it starts the browser sign-in in the background and relays a URL and code into the chat for you to approve, usually a single click. Once you're signed in, it registers every project you can reach on that stack and verifies the result. Anything already done is skipped, so re-running it is safe. + +The other four clients cannot run it. Install the CLI and connect your project in a terminal first, then add the plugin from your client's UI. + + + +## Set up your client + +:::tip[You sign in once per machine] +Every tool reads the same local config, so a project you connected from one client is already there in the others. Run `kbagent project list` first. If your project is already listed, the terminal steps below are done, and step 1 of your client's section is too. +::: + +### First, in a terminal + +Every client except Claude Code starts here. + +1. [Install the CLI](/cli/getting-started/) for your operating system. +2. [Connect your project](/cli/getting-started/#step-2--connect-your-project). Sign in with `kbagent auth login`, or register it with a Storage API token. +3. Check what you have: `kbagent doctor` + +Use a real terminal for all three. `auth login` opens a browser you have to finish at, and a tokenless `project add` prompts with hidden input, which needs a TTY an agent's tool-run shell does not have. + +`kbagent doctor` sits here, with the steps it checks. It looks for a plugin under `~/.claude/plugins/cache`, so it cannot see a Cursor, VS Code or ChatGPT app install, and it cannot confirm the plugin half of your setup. + +Then follow your client below. + +### Claude Code + +Claude Code is already a terminal, so it can do the whole setup itself. + +1. Start it: `claude` +2. Add the marketplace: `/plugin marketplace add keboola/ai-kit` +3. Install the plugin: `/plugin install kbagent@keboola-claude-kit` +4. Run the setup: `/kbagent:setup https://connection.keboola.com` + +5. Ask it: `kbagent list my projects` + +### Claude Desktop + +Claude Desktop has no slash commands in its chat. `/plugin` answers "/plugin isn't available in this environment", and both `/kbagent:setup` and `/kbagent` answer "Unknown command". The plugin goes in through the UI instead. + +1. Do the terminal steps above. +2. Open **Customise → Plugins → Add → Add from marketplace** and paste `keboola/ai-kit`. The short form works here. +3. Find the card titled **kbagent** and click its plus. + +4. Ask in the chat: `kbagent list my projects` + +Type that last one without a leading slash. `/kbagent …` fails here. + +### Cursor + +1. Do the terminal steps above. +2. Open **Customise → Browse Marketplace → Add Marketplace → Import from GitHub**. +3. Paste the full URL as the repository: `https://github.com/keboola/ai-kit` +4. Find the `kbagent` row under **Keboola Ai Kit** and press **Add**. + +5. Ask in the chat: `kbagent list my projects` + +:::caution[Cursor needs the full URL] +The short `keboola/ai-kit` form that works in Claude Code and Claude Desktop is rejected here with `[invalid_argument] Error`. Paste `https://github.com/keboola/ai-kit`. +::: + +### VS Code + +VS Code's Copilot reads the Claude plugin format (`.claude-plugin/plugin.json` and `.claude-plugin/marketplace.json`), so AI Kit works there unchanged. + +1. Do the terminal steps above. +2. Open the Command Palette (`⇧⌘P`, or `Ctrl+Shift+P` on Windows) and run **Chat: Install Plugin from Source**. Several palette entries start with "Install", so match the whole name. +3. Paste `https://github.com/keboola/ai-kit` as the source, then confirm the Trust prompt. VS Code asks because a plugin can run code. +4. Pick `kbagent` from the picker. + +5. Open the Chat panel (`⌃⌘I`, or `Ctrl+Alt+I` on Windows) and ask: `kbagent list my projects` + +### ChatGPT app + +The ChatGPT app, formerly Codex, accepts `.claude-plugin/marketplace.json` and `git-subdir` sources, which is how AI Kit publishes the plugin. Nothing Codex-specific is needed. + +1. Do the terminal steps above. +2. In **Settings**, turn on **Developer mode**. Without it the app has nowhere to add a marketplace from. Its warning mentions unverified connectors and permanent data loss. Here that means the plugin runs the CLI under your own login, so it can do whatever your token can. +3. In the plugins view, open **Add → Add a marketplace** and paste `https://github.com/keboola/ai-kit` as the source. Leave **Git ref** and **Sparse paths** empty. The grey text in them is placeholder, and `plugins/codex` in particular reads like a value you should keep. +4. Switch to the **Personal** tab, where an added marketplace is listed, and install `kbagent`. + +5. Ask in the chat: `kbagent list my projects` + +The same two steps work from a shell: + +```bash +codex plugin marketplace add https://github.com/keboola/ai-kit +codex plugin add kbagent@keboola-claude-kit +``` + +### Plain terminal + +No AI client involved: [install the CLI](/cli/getting-started/), [connect your project](/cli/getting-started/), verify with `kbagent doctor`, and list what you can reach with `kbagent project list`. + + ## The `context` reference @@ -60,11 +160,11 @@ export KBAGENT_CONVERSATION_ID="" ## How it fits with the other AI tools - **kbagent** — the agent's hands on your projects from the terminal, with sandboxing. -- **[MCP server](/ai/mcp-server/)** — direct tool calls over MCP; kbagent can also call MCP tools via `kbagent tool`. +- **[MCP server](/ai/mcp-server/)** — direct tool calls over MCP. - **[AI Kit](/ai/ai-kit/)** — coding-assistant plugins for building Keboola components and apps. - **[Kai](/kai/)** — the in-product assistant; `kbagent kai ask -m "why did last night's load fail?"` puts the same assistant in your shell *(beta)*. - + --- diff --git a/src/content/docs/cli/getting-started.mdx b/src/content/docs/cli/getting-started.mdx index 6d284d867..42bd2e674 100644 --- a/src/content/docs/cli/getting-started.mdx +++ b/src/content/docs/cli/getting-started.mdx @@ -18,8 +18,6 @@ This walkthrough takes you from nothing to browsing a real project with [kbagent Pick your operating system: -{/* VERIFY(owner: Padak): provisional canon — install script marked Recommended on macOS/Linux (matches the landing page and README), uv as the alternative. Flip the labels if uv-first is preferred. */} - **Recommended** — the prebuilt wheel via the install script: @@ -48,11 +46,36 @@ Pick your operating system: ``` - **Recommended** — [uv](https://docs.astral.sh/uv/) in PowerShell, the one supported native path on Windows (install uv first if you don't have it): + The `curl … | sh` one-liner **cannot run on Windows**. Windows has shipped `curl.exe` since Windows 10 1803, but it has no `sh`, so the command fails with `'sh' is not recognized` in both `cmd` and PowerShell. Installing Git for Windows does not fix that. Its installer only puts `C:\Program Files\Git\cmd` on `PATH`, while `sh.exe` lives in `usr\bin`. + + **Recommended**: PowerShell, no POSIX shell needed. This fetches the same wheel the install script would have: ```powershell - uv tool install "git+https://github.com/keboola/cli" + winget install --id astral-sh.uv -e + $ver = (Invoke-RestMethod "https://api.github.com/repos/keboola/cli/releases/latest").tag_name.TrimStart('v') + uv tool install --force "keboola-cli[server] @ https://github.com/keboola/cli/releases/download/v$ver/keboola_cli-$ver-py3-none-any.whl" + uv tool update-shell ``` + + Skip the first line if you already have [uv](https://docs.astral.sh/uv/). That line installs uv itself. The whole block works in the in-box Windows PowerShell 5.1, so you don't need PowerShell 7. + + The second line calls the GitHub API without a token. If you are rate-limited or behind a proxy that blocks it, `$ver` comes back empty and the third line asks uv for a URL that does not exist, which surfaces as a 404. Check with `echo $ver`. If it is empty, read the version off the [releases page](https://github.com/keboola/cli/releases/latest) and set it by hand, e.g. `$ver = "0.91.0"`. + + Then **open a new shell**. `uv tool update-shell` edits the persisted `PATH`, and the session you ran it in will not see the change. + + Two alternatives: + + - **Already have Git for Windows?** Run the documented script through its bash: + + ```powershell + & "C:\Program Files\Git\bin\bash.exe" -lc "curl -LsSf https://raw.githubusercontent.com/keboola/cli/main/install.sh | sh" + ``` + + - **No Python on the machine?** Every release ships a self-contained build. Download `keboola-cli2__windows_amd64.zip` from the [releases page](https://github.com/keboola/cli/releases/latest), unpack it, and put the folder on `PATH`. It carries its own interpreter and needs neither Python nor uv. + + :::caution[Not via a package manager] + There is no WinGet package for kbagent itself, and the Chocolatey package lags far behind the current release. Use one of the three paths above. + ::: @@ -62,9 +85,13 @@ Confirm it's on your `PATH` (all platforms): ```console $ kbagent --version -kbagent v0.66.0 +kbagent v0.91.0 ``` -{/* Verified locally 2026-07-13 (macOS): kbagent v0.66.0 via uv. Windows/PowerShell uv path + install.sh being bash-only confirmed by Padak (review, v0.66.1). */} +{/* Verified locally 2026-07-13 (macOS) via uv; version sample tracks the vendored command reference. Windows block (winget uv + release wheel + `uv tool update-shell`, Git-bash and self-contained-zip alternatives, no WinGet package, stale Chocolatey) copied from the "Windows" section of keboola/cli's README, which is the source of truth for it. */} + +:::note +On macOS and Linux the install script does not touch the `PATH` of the shell that ran it. If `kbagent: command not found` comes back right afterwards, the install worked and your `PATH` has not caught up. Run `source $HOME/.local/bin/env`, or open a new terminal. +::: :::tip On any platform you can also run kbagent through [uvx](https://docs.astral.sh/uv/) without installing (`uvx --from 'git+https://github.com/keboola/cli' kbagent …`) — handy for a one-off or in CI. @@ -72,22 +99,56 @@ On any platform you can also run kbagent through [uvx](https://docs.astral.sh/uv ## Step 2 — Connect your project -For one project you need a Storage API token and your stack URL. To create the token, open your project in the browser and go to **Project Settings → API Tokens → New Token**: give it a description, pick **Full Access** (you can scope it down later), and click **Create** — then copy the token right away, it's shown only once. +Register a project with the CLI so you (or your agent) can reach it. There are two ways in: sign in through the browser, or hand kbagent a Storage API token. + +### Sign in through the browser + +At your own keyboard, this is the shortest route. It needs no token. + +First sign in to the stack: -![The New Token form in Keboola Project Settings: a description field, an Expires In selector, and Full Access selected for both Files and Components & Buckets, with the Create button in the corner](/cli/token-create.png) -{/* Screenshot: live capture of /admin/projects/264/tokens-settings/new-token, demo project 264, 2026-07-24. */} +```bash +kbagent auth login --stack https://connection.keboola.com +``` + +That opens a browser, and you finish the sign-in there. On a machine with no browser it prints a code you type into a page on another device instead. + +Then pick which projects to register locally: ```bash -kbagent project add --project prod \ - --url https://connection.keboola.com --token YOUR_TOKEN +kbagent auth register-projects ``` -`prod` is an **alias** you pick — you'll use it with `--project` later, or set it as the default with `kbagent project use prod`. +It shows a picker of every project the account can reach and registers the ones you choose. Add `--all` to register all of them without the picker. + +Both commands need a real terminal. There is no headless path through `auth login`. + +### Or register one project with a token + +```bash +kbagent project add --project prod --url https://connection.keboola.com --token YOUR_TOKEN +``` + +`project add` is the static-token route. It reads the token from `--token`, or from `KBC_TOKEN` if you'd rather keep it off the command line. With neither, it prompts with hidden input, so the value never lands in your shell history. That prompt needs a TTY an agent's tool-run shell does not have. + +Swap the URL for your own stack. `prod` is an **alias** you pick. You use it with `--project` later, or set it as the default with `kbagent project use prod`. Add as many projects as you want. + +You sign in once per machine. Every tool reads the same local config, so a project you connected here is already there in Claude Code, Cursor, VS Code and the rest. `kbagent project list` shows what is registered, and its `Auth` column says `session` or `static` per project. + +### When you need a token + +CI, containers, cron, and anything else running unattended need a Storage API token, because `auth login` cannot run there. A few features need a static token rather than a browser session too: Kai, data apps, the semantic layer, streams, and the Python SDK. + +**These docs cannot create a token for you.** Open your project in the Keboola UI and go to **Settings → Developer settings → Agentic CLI**. That page builds the exact `kbagent project add` command for your project and your stack, and it can include a read-only token in it. Copy it from there. + +If you don't see **Agentic CLI** in your project's settings, create the token yourself under **Settings → API Tokens** and pass it as `--token`. [API tokens](/management/project/tokens/) covers that screen. + +{/* Step 2's static-token half mirrors the Agentic CLI surface in the product UI (Settings → Developer settings → Agentic CLI): the same `kbagent project add --project --url ` and the read-only token that page can put into the command. That page is feature-flagged, hence the API Tokens fallback. Browser-login half (`auth login` + `auth register-projects`, `--all`, the device-code path, the `Auth` column, no headless path) and the static-token-only features (Kai, data apps, semantic layer, streams, Python SDK) from keboola/cli docs/auth.md on main. */} A Storage API token is enough for browsing. Some commands need an admin (master) token, e.g. creating branches or workspaces needs admin privileges. Token types and scoping options are covered in [API tokens](/management/project/tokens/). :::caution -kbagent only talks to your Keboola stack, but the config directory stores project credentials — treat it as sensitive, and never paste tokens into shared terminals or commit them. +kbagent only talks to your Keboola stack, but the config directory stores project credentials. Treat it as sensitive, and never paste tokens into shared terminals or commit them. ::: ## Step 3 — Verify the connection @@ -134,7 +195,7 @@ kbagent search "shopify" ## Connect more projects -**Several projects** — register each with its own Storage token (`project add`), or bulk-onboard with a Manage API token: +**Several projects** — register each one with `project add`, or bulk-onboard with a Manage API token: ```bash KBC_MANAGE_API_TOKEN=xxx kbagent --allow-env-manage-token \ diff --git a/src/content/docs/cli/index.md b/src/content/docs/cli/index.md index b1547d167..84013bb15 100644 --- a/src/content/docs/cli/index.md +++ b/src/content/docs/cli/index.md @@ -12,10 +12,6 @@ It's just as comfortable in your own hands: one tool for every project from the -:::caution[Beta] -kbagent is in beta. Commands and output formats may still change. -::: - :::note kbagent is a different tool from the legacy **Keboola as Code** CLI (`kbc`) documented on [developers.keboola.com/cli](https://developers.keboola.com/cli/). That tool is still supported for now; new command-line work should use kbagent. @@ -35,12 +31,21 @@ kbagent is a different tool from the legacy **Keboola as Code** CLI (`kbc`) docu ## In one minute +Install it (recommended on macOS/Linux; Windows has its own path under [Get started](/cli/getting-started/)): + ```bash -# install (recommended on macOS/Linux; Windows → Get started) curl -LsSf https://raw.githubusercontent.com/keboola/cli/main/install.sh | sh -# connect a project -kbagent project add --project prod --url https://connection.keboola.com --token YOUR_TOKEN -# explore +``` + +Connect your projects. A browser opens, and you finish the sign-in there: + +```bash +kbagent auth login --stack https://connection.keboola.com --register-projects +``` + +Then explore: + +```bash kbagent doctor kbagent job list --limit 5 kbagent search "customer_id" @@ -50,7 +55,7 @@ A minute in, `doctor` has confirmed the connection, you've seen your last five j ## Why a CLI (and not just the UI)? -The UI is great for one project, one change at a time. kbagent is for the things the UI makes hard: doing the same thing across **many projects at once**, putting your configuration **under version control**, wiring Keboola into **CI/CD**, and letting an **AI agent** do the work while you keep approval of anything destructive. It talks only to your Keboola stacks, stores its connections locally, and never needs a browser. +The UI is great for one project, one change at a time. kbagent is for the things the UI makes hard: doing the same thing across **many projects at once**, putting your configuration **under version control**, wiring Keboola into **CI/CD**, and letting an **AI agent** do the work while you keep approval of anything destructive. It talks only to your Keboola stacks and stores its connections locally. Signing in opens a browser once; after that every command runs in your terminal, and a Storage API token skips the browser entirely. ## This section @@ -59,7 +64,7 @@ The pages read in order, from first run to deep reference: 1. **[Get started](/cli/getting-started/)** — install (macOS / Linux / Windows), connect a project, run your first commands. 2. **[How it works](/cli/concepts/)** — the connection model, multi-project, dev branches, GitOps sync, and the safety firewall. 3. **[How-to guides](/cli/workflows/)** — task recipes: onboard an org, dev-branch workflow, GitOps sync, audits, CI/CD tokens, encryption. -4. **[Use with AI agents](/cli/for-agents/)** — the Claude Code plugin, `kbagent context`, and sandboxing an agent. +4. **[Use with AI agents](/cli/for-agents/)** — the kbagent plugin and its setup in each AI client, `kbagent context`, and sandboxing an agent. 5. **[Command reference](/cli/commands/)** — every command group, global flags, JSON output, and error codes. 6. **[Web UI](/cli/web-ui/)** — the optional local browser dashboard. diff --git a/src/content/docs/cli/troubleshooting.md b/src/content/docs/cli/troubleshooting.md index cae7eb41d..8aa4e6e05 100644 --- a/src/content/docs/cli/troubleshooting.md +++ b/src/content/docs/cli/troubleshooting.md @@ -10,7 +10,7 @@ description: 'The most common kbagent CLI failures and their fixes — install a The failures you're most likely to hit with [kbagent](/cli/), each with its cause and fix. With `--json`, failures carry a stable machine-readable `code` — it's shown here where one applies, and the [full catalogue](https://github.com/keboola/cli/blob/main/docs/error-codes.md) lives in the CLI repo. -`kbagent doctor` is the universal first step: it checks your config, connections, and version, and often names the problem outright. `kbagent doctor --fix` repairs what it safely can. +`kbagent doctor` is the universal first step: it checks your config, connections, and version, and often names the problem outright. It reports what it finds; the fixes below are the manual follow-up. @@ -120,7 +120,7 @@ kbagent changelog ## Still stuck? -- `kbagent doctor --fix` — automated checks and safe repairs. +- `kbagent doctor` — health checks on your config and project connectivity. - `kbagent context` — the full command reference for your installed version. - The [error-code catalogue](https://github.com/keboola/cli/blob/main/docs/error-codes.md) — every `--json` `code` with its meaning. - [Command reference](/cli/commands/) — flags and groups overview. diff --git a/src/content/docs/components/branches/index.md b/src/content/docs/components/branches/index.md index 9cf9abe51..55c563fa8 100644 --- a/src/content/docs/components/branches/index.md +++ b/src/content/docs/components/branches/index.md @@ -1,134 +1,134 @@ ---- -title: Development Branches -slug: 'components/branches' ---- - - - -*If you already know how development branches work in general and want to create and start using your first branch, -go to our [Getting Started tutorial](/getting-started/branches/).* - -Development Branches allow you to modify [component configurations](/components/) without interfering with running -configurations or entire [automated pipelines](/flows/). They are ideal to use when making bigger changes -to a project or when you need to be extra careful about performing your changes safely. - -To give an example, let's say that you have an ordinary flow that extracts, transforms and writes data -to a target system, and you need to remove a column from the source. To do that, you must modify several configurations, -and ideally, also perform a dry run to check that the data in the target system is correct. However, modifying a pipeline -that runs, e.g., every ten minutes, is difficult without an outage of the pipeline. Development Branches are designed -to help in such situations. - -## How Branches Work - -When you create a development branch in your project, you obtain an exact copy of the project and all its current -configurations. You can then modify these configurations without ever touching the original ones in production, -and these will keep running in flows. - -When you run a configuration in a branch, it can **read** the [tables](/storage/tables/) and [files](/storage/files/) -from Storage as if it were a normal configuration. However, when your branch configuration attempts to **write** data -(tables or files), the data is written to the branch's isolated storage layer. This means that production data and branch data -are completely separated. There is no need to duplicate your entire project's data when creating a new branch. - -## Branched Storage - -Branched Storage is an improved storage isolation model for development branches. Instead of cluttering your project -with prefixed bucket names (like `in.c-1234-bucket`), each branch gets its own fully isolated storage namespace. -Production data is never touched, and no data is copied up front — a copy is created only when you actually write to or modify a table within the branch. - -:::tip -Branched Storage is available for projects running on Snowflake. If your project uses BigQuery, the classic prefix-based model still applies — Branched Storage support for BigQuery is coming. -::: - -### Why It Matters - -Without Branched Storage, every write in a branch produced new buckets with prefixed names that were visible in -production Storage and cluttered the namespace. You had to be careful about what you ran and where. - -With Branched Storage: - -- **Production is safe** — writes in a branch never affect production data. -- **No data duplication up front** — creating a branch is instant and doesn't copy your storage. Tables are only materialized when you write to them. -- **Reads are transparent** — if a table hasn't been modified in the branch, you're reading live production data, with no extra cost. -- **Clean Storage** — the branch has its own storage namespace. No prefixed buckets visible in production. - -![Screenshot - Branched Storage in Storage UI](/components/branches/branched_storage.png) - -### Enabling Branched Storage - -Branched Storage is enabled per project in **Project Settings**. Look for the **Branched storage** toggle under the Features section. - -![Screenshot - Branched Storage Toggle](/components/branches/feature-branched-storage.png) - -Once enabled, all new branches in that project will automatically use the isolated storage model. - -### How It Behaves in Practice - -When you create a branch and run a job: - -1. **Reading** from a table that hasn't been modified → the branch reads directly from production. Nothing is copied. -2. **Writing** to a table for the first time → the table is cloned into the branch's isolated storage. All subsequent reads and writes for that table within the branch use this isolated copy. -3. **Production is never touched** — regardless of how many times you write in a branch. - -When you delete or merge the branch, the branched storage is cleaned up accordingly. - -## Data Pipelines in Branches - -In production, you might have a data source connector that extracts data into a bucket `in.c-requests`, and a transformation -that reads from it and writes results to `out.c-visits`. - -When you switch to a branch on a **Snowflake project with Branched Storage enabled**, no data is copied immediately. -The branch reads from production Storage until you run a job that writes data — at that point, only the affected tables -are materialized in the branch's own storage. Your production `out.c-visits` remains untouched throughout. - -This allows you to test the entire pipeline with real data, in complete isolation from production, without duplicating -all storage content at branch creation. - -## Creating a Branch - -If you have your configurations ready in production and want to create a branch to test some changes, click on your project's name -at the top of the screen. Then click on the green icon **New** displayed next to your project's name. - -![Screenshot - Create Development Branch](/getting-started/branches/figures/08-create-dev-branch.png) - -Name your new branch and click **Create Development Branch** to open it. - -![Screenshot - Name Development Branch](/getting-started/branches/figures/09-name-dev-branch.png) - -The branch will appear right below the name of your production project. - -![Screenshot - Created Development Branch](/getting-started/branches/figures/10-dev-branch-created.png) - -Now you can start modifying your configurations, run them, and analyze the results. - -If you want to learn more about working in a branch, follow our [tutorial](/getting-started/branches/). - -## Closing a Branch - -Before you merge your development branch back to production, check a detailed [diff of the configuration changes](/getting-started/branches/project-diff/). - -You can end your branch's lifecycle in two ways: - -- **Deleting** — if you do not wish to use the changes you've made and want to simply discard them. The data associated with the branch is discarded when the branch is deleted. -- [**Merging into production**](/getting-started/branches/merge-to-production/) — all changes in the configurations are brought back to the respective production configurations. All the changes are applied at once (after you approve them) and produce new [versions](/components/#configuration-versions) of the respective configurations. The branch can be either deleted or kept for further reference after merging. Projects with [**Branches 2.0**](/components/branches/merge-requests/) enabled merge through a [merge request](/components/branches/merge-requests/), which lets you review the configuration diff and, if the project requires it, collect approvals before the changes reach production. - -***Important:** All of this happens within the same project, enabling collaboration with other project members on the modifications.* - -## Component Considerations - -Certain components are not allowed to run in development branches. There are following special cases where components' functionality is limited in the development branches. - -### Working with External Resources - -Some components, like writers, can write to a destination that is external to Keboola. Those components' -configs are first marked as *unsafe* in development branch. - -You will not be able to run an unsafe config. You need to first observe the config and verify that it's either OK to -write to the destination, or change the destination accordingly. - -### OAuth Authorized Components - -Components using OAuth do not allow authorizing nor changing the OAuth in a development branch. The OAuth authorization tokens are shared with production so changing them might break the production pipeline. - -***** - -***Important:** Development branches are for development and testing only, so setting up status notifications on Flows is not supported.* +--- +title: Development Branches +slug: 'components/branches' +--- + + + +*If you already know how development branches work in general and want to create and start using your first branch, +follow the [Branches Tutorial](/components/branches/tutorial/).* + +Development Branches allow you to modify [component configurations](/components/) without interfering with running +configurations or entire [automated pipelines](/flows/). They are ideal to use when making bigger changes +to a project or when you need to be extra careful about performing your changes safely. + +To give an example, let's say that you have an ordinary flow that extracts, transforms and writes data +to a target system, and you need to remove a column from the source. To do that, you must modify several configurations, +and ideally, also perform a dry run to check that the data in the target system is correct. However, modifying a pipeline +that runs, e.g., every ten minutes, is difficult without an outage of the pipeline. Development Branches are designed +to help in such situations. + +## How Branches Work + +When you create a development branch in your project, you obtain an exact copy of the project and all its current +configurations. You can then modify these configurations without ever touching the original ones in production, +and these will keep running in flows. + +When you run a configuration in a branch, it can **read** the [tables](/storage/tables/) and [files](/storage/files/) +from Storage as if it were a normal configuration. However, when your branch configuration attempts to **write** data +(tables or files), the data is written to the branch's isolated storage layer. This means that production data and branch data +are completely separated. There is no need to duplicate your entire project's data when creating a new branch. + +## Branched Storage + +Branched Storage is an improved storage isolation model for development branches. Instead of cluttering your project +with prefixed bucket names (like `in.c-1234-bucket`), each branch gets its own fully isolated storage namespace. +Production data is never touched, and no data is copied up front — a copy is created only when you actually write to or modify a table within the branch. + +:::tip +Branched Storage is available for projects running on Snowflake. If your project uses BigQuery, the classic prefix-based model still applies — Branched Storage support for BigQuery is coming. +::: + +### Why It Matters + +Without Branched Storage, every write in a branch produced new buckets with prefixed names that were visible in +production Storage and cluttered the namespace. You had to be careful about what you ran and where. + +With Branched Storage: + +- **Production is safe** — writes in a branch never affect production data. +- **No data duplication up front** — creating a branch is instant and doesn't copy your storage. Tables are only materialized when you write to them. +- **Reads are transparent** — if a table hasn't been modified in the branch, you're reading live production data, with no extra cost. +- **Clean Storage** — the branch has its own storage namespace. No prefixed buckets visible in production. + +![Screenshot - Branched Storage in Storage UI](/components/branches/branched_storage.png) + +### Enabling Branched Storage + +Branched Storage is enabled per project in **Project Settings**. Look for the **Branched storage** toggle under the Features section. + +![Screenshot - Branched Storage Toggle](/components/branches/feature-branched-storage.png) + +Once enabled, all new branches in that project will automatically use the isolated storage model. + +### How It Behaves in Practice + +When you create a branch and run a job: + +1. **Reading** from a table that hasn't been modified → the branch reads directly from production. Nothing is copied. +2. **Writing** to a table for the first time → the table is cloned into the branch's isolated storage. All subsequent reads and writes for that table within the branch use this isolated copy. +3. **Production is never touched** — regardless of how many times you write in a branch. + +When you delete or merge the branch, the branched storage is cleaned up accordingly. + +## Data Pipelines in Branches + +In production, you might have a data source connector that extracts data into a bucket `in.c-requests`, and a transformation +that reads from it and writes results to `out.c-visits`. + +When you switch to a branch on a **Snowflake project with Branched Storage enabled**, no data is copied immediately. +The branch reads from production Storage until you run a job that writes data — at that point, only the affected tables +are materialized in the branch's own storage. Your production `out.c-visits` remains untouched throughout. + +This allows you to test the entire pipeline with real data, in complete isolation from production, without duplicating +all storage content at branch creation. + +## Creating a Branch + +If you have your configurations ready in production and want to create a branch to test some changes, click on your project's name +at the top of the screen. Then click on the green icon **New** displayed next to your project's name. + +![Screenshot - Create Development Branch](/components/branches/tutorial/figures/08-create-dev-branch.png) + +Name your new branch and click **Create Development Branch** to open it. + +![Screenshot - Name Development Branch](/components/branches/tutorial/figures/09-name-dev-branch.png) + +The branch will appear right below the name of your production project. + +![Screenshot - Created Development Branch](/components/branches/tutorial/figures/10-dev-branch-created.png) + +Now you can start modifying your configurations, run them, and analyze the results. + +If you want to learn more about working in a branch, follow the [Branches Tutorial](/components/branches/tutorial/). + +## Closing a Branch + +Before you merge your development branch back to production, check a detailed [diff of the configuration changes](/components/branches/tutorial/project-diff/). + +You can end your branch's lifecycle in two ways: + +- **Deleting** — if you do not wish to use the changes you've made and want to simply discard them. The data associated with the branch is discarded when the branch is deleted. +- [**Merging into production**](/components/branches/tutorial/merge-to-production/) — all changes in the configurations are brought back to the respective production configurations. All the changes are applied at once (after you approve them) and produce new [versions](/components/#configuration-versions) of the respective configurations. The branch can be either deleted or kept for further reference after merging. Projects with [**Branches 2.0**](/components/branches/merge-requests/) enabled merge through a [merge request](/components/branches/merge-requests/), which lets you review the configuration diff and, if the project requires it, collect approvals before the changes reach production. + +***Important:** All of this happens within the same project, enabling collaboration with other project members on the modifications.* + +## Component Considerations + +Certain components are not allowed to run in development branches. There are following special cases where components' functionality is limited in the development branches. + +### Working with External Resources + +Some components, like writers, can write to a destination that is external to Keboola. Those components' +configs are first marked as *unsafe* in development branch. + +You will not be able to run an unsafe config. You need to first observe the config and verify that it's either OK to +write to the destination, or change the destination accordingly. + +### OAuth Authorized Components + +Components using OAuth do not allow authorizing nor changing the OAuth in a development branch. The OAuth authorization tokens are shared with production so changing them might break the production pipeline. + +***** + +***Important:** Development branches are for development and testing only, so setting up status notifications on Flows is not supported.* diff --git a/src/content/docs/getting-started/branches/files-in-branch.md b/src/content/docs/components/branches/tutorial/files-in-branch.md similarity index 72% rename from src/content/docs/getting-started/branches/files-in-branch.md rename to src/content/docs/components/branches/tutorial/files-in-branch.md index 6d27b694d..1128c0d49 100644 --- a/src/content/docs/getting-started/branches/files-in-branch.md +++ b/src/content/docs/components/branches/tutorial/files-in-branch.md @@ -1,22 +1,23 @@ --- title: 'Working with Files in a Branch' -slug: 'getting-started/branches/files-in-branch' +slug: 'components/branches/tutorial/files-in-branch' description: 'See how File Storage behaves inside a development branch and how branch-created files are tagged and kept apart from production files.' redirect_from: - /tutorial/branches/files-in-branch/ + - /getting-started/branches/files-in-branch/ --- Now that you have created a branch from configurations in production and tested how -[tables behave in development branches](/getting-started/branches/tables-in-branch/), let's do the same with +[tables behave in development branches](/components/branches/tutorial/tables-in-branch/), let's do the same with [files](/storage/files/). ## Run Transformation in Branch In your development branch, go to **Transformations**, select the transformation `Sample Python transformation` and run it. -![Screenshot - Run Transformation in Development Branch](/getting-started/branches/figures/python-branch-overview.png) +![Screenshot - Run Transformation in Development Branch](/components/branches/tutorial/figures/python-branch-overview.png) ## File Tags When the job finishes, go to **Storage -- Files**. You can see that the `demoFile.txt` was created but is was not @@ -24,7 +25,7 @@ assigned the tag `demoOutput`. It was assigned the tag `1835-demoOutput` instead this branch. You can also see the ID in the URL. This is how the development branch files are distinguished from the files that were created in the production environment. -![Screenshot - Development Branch Output](/getting-started/branches/figures/12-dev-branch-output.png) +![Screenshot - Development Branch Output](/components/branches/tutorial/figures/12-dev-branch-output.png) ## Change Configuration Now it is time to make the changes to the transformation and test them. Go to **Transformations**, find `Sample Python @@ -40,12 +41,12 @@ print("Output written to demoFile.txt") ``` -![Screenshot - Edit Code Block](/getting-started/branches/figures/python-branch-change-code.png) +![Screenshot - Edit Code Block](/components/branches/tutorial/figures/python-branch-change-code.png) The transformation will do exactly the same as before, but it will add `Output written to demoFile.txt` message to the output log. Save the code block and run the transformation again. You can see in the job log that it did indeed output the log message. -![Screenshot - Job Log](/getting-started/branches/figures/14-jobs-log.png) +![Screenshot - Job Log](/components/branches/tutorial/figures/14-jobs-log.png) The file was uploaded to Storage again. @@ -56,13 +57,13 @@ affect the production transformation in any way. Go to **Transformations**, open the `Sample Python transformation` transformation and check the code block to see that there is still the original version without the log output that we created at the beginning. It was not overwritten by the changes we made in the development branch. -![Screenshot - Check Variable In Production](/getting-started/branches/figures/20-check-block1.png) +![Screenshot - Check Variable In Production](/components/branches/tutorial/figures/20-check-block1.png) To make sure nothing has changed, run the transformation in production again. When it finishes, go to **Storage -- Files**. You can see that `demoFile.txt` was created. Because it ran in production, it was assigned the tag `demoOutput` without any prefix. -![Screenshot - Storage File in Production](/getting-started/branches/figures/21-storage-files-prod.png) +![Screenshot - Storage File in Production](/components/branches/tutorial/figures/21-storage-files-prod.png) This concludes the file manipulation part of this tutorial. You examined how files behave in branches. Now you can -examine the [changes you have made in the branch](/getting-started/branches/project-diff). +examine the [changes you have made in the branch](/components/branches/tutorial/project-diff). diff --git a/src/content/docs/components/branches/tutorial/index.md b/src/content/docs/components/branches/tutorial/index.md new file mode 100644 index 000000000..4b5f9d99d --- /dev/null +++ b/src/content/docs/components/branches/tutorial/index.md @@ -0,0 +1,32 @@ +--- +title: 'Branches Tutorial' +slug: 'components/branches/tutorial' +description: "Change a running project safely: work in a development branch, see the project diff, and merge your changes back into production." +redirect_from: + - /tutorial/branches/ + - /getting-started/branches/ +--- + + +Development branches let you modify [component configurations](/components/) without interfering +with the running configurations or entire [automated pipelines](/flows/). This tutorial walks the +whole cycle end to end — you build production configurations, change them inside a branch, review +the diff, and merge back. + +[Development Branches](/components/branches/) explains how they work first; come back here to try +it. Projects with **Branches 2.0** enabled merge through a +[merge request](/components/branches/merge-requests/) instead of the direct merge shown in part 3. + +* Part 1 -- Preparing production configurations: + * [Preparing table manipulating configurations](/components/branches/tutorial/prepare-tables/) + * [Preparing file manipulating configurations](/components/branches/tutorial/prepare-files/) +* Part 2 -- Working in a branch: + * [Working with tables in a branch](/components/branches/tutorial/tables-in-branch) + * [Working with files in a branch](/components/branches/tutorial/files-in-branch) +* Part 3 -- Merging branches: + * [Project diff](/components/branches/tutorial/project-diff/) + * [Merge to production](/components/branches/tutorial/merge-to-production/) + +:::caution[Public Beta] +This feature is currently in public beta. Please provide feedback using the feedback button in your project. +::: diff --git a/src/content/docs/getting-started/branches/merge-to-production.md b/src/content/docs/components/branches/tutorial/merge-to-production.md similarity index 78% rename from src/content/docs/getting-started/branches/merge-to-production.md rename to src/content/docs/components/branches/tutorial/merge-to-production.md index a17f121a1..e30df4db7 100644 --- a/src/content/docs/getting-started/branches/merge-to-production.md +++ b/src/content/docs/components/branches/tutorial/merge-to-production.md @@ -1,14 +1,15 @@ --- title: 'Merge to Production' -slug: 'getting-started/branches/merge-to-production' +slug: 'components/branches/tutorial/merge-to-production' description: 'Merge a development branch back into production — all at once or configuration by configuration — and verify the result.' redirect_from: - /tutorial/branches/merge-to-production/ + - /getting-started/branches/merge-to-production/ --- -After you have [checked your changes in the branch](/getting-started/branches/project-diff/), you can merge the branch +After you have [checked your changes in the branch](/components/branches/tutorial/project-diff/), you can merge the branch back to production. There are two ways to merge your changes to production, depending on whether you want to merge them all or just their subset: @@ -24,7 +25,7 @@ transformation), you now have a good opportunity to test both approaches. First, let's merge only the subset of configurations related to the `Sample Python transformation`. Examine the project diff further. -![Screenshot - Project diff](/getting-started/branches/figures/merge-python-checkbox.png) +![Screenshot - Project diff](/components/branches/tutorial/figures/merge-python-checkbox.png) You can see that there are checkboxes to the left of each configuration in the list. The configurations that have the checkbox checked will be merged. Uncheck all the checkboxes except the one near the `Sample Python transformation`. @@ -35,27 +36,27 @@ after merge.* is not checked. Only then click the **Merge** button. ***Note:** If you merged the branch with the checkbox checked, you will need to recreate the whole branch in the next step.* -![Screenshot - Merge dialog](/getting-started/branches/figures/merge-python-dialog.png) +![Screenshot - Merge dialog](/components/branches/tutorial/figures/merge-python-dialog.png) When you start the merge, a progress bar will show up informing you of the progress of the merge. After the merge is finished, you will see only two changed configurations in your branch. The Python transformation configuration no longer differs. -![Screenshot - Match Change from Production](/getting-started/branches/figures/partially-merged-branch.png) +![Screenshot - Match Change from Production](/components/branches/tutorial/figures/partially-merged-branch.png) Switch to production, and examine the Python transformation configuration. Notice that a new version has been created with the merge message as the description of the change. -![Screenshot - Merged change](/getting-started/branches/figures/merge-python-in-prod.png) +![Screenshot - Merged change](/components/branches/tutorial/figures/merge-python-in-prod.png) If you examine the code block, you will see that the change from the branch is there. -![Screenshot - Merged change](/getting-started/branches/figures/merge-python-in-prod-2.png) +![Screenshot - Merged change](/components/branches/tutorial/figures/merge-python-in-prod-2.png) If you go to the **Storage** section, you will see that the branch buckets are still available if you toggle the switch to show development branch buckets. However, they cannot be used by the production configurations. -![Screenshot - Production storage](/getting-started/branches/figures/merge-python-prod-storage.png) +![Screenshot - Production storage](/components/branches/tutorial/figures/merge-python-prod-storage.png) ## Full Merge @@ -64,17 +65,17 @@ respective checkboxes checked. Then click **Merge to production**. Make sure tha branch after merge.* is checked. Fill in the merge message: `Merge the Bitcoin transformation and HTTP data source connector`. Then click **Merge**. -![Screenshot - Merge dialog](/getting-started/branches/figures/merge-snflk-dialog.png) +![Screenshot - Merge dialog](/components/branches/tutorial/figures/merge-snflk-dialog.png) The merge will take slightly longer as the whole branch is being deleted. Afterwards, you will be redirected back to production. -![Screenshot - Branch deleted](/getting-started/branches/figures/branch-deleted.png) +![Screenshot - Branch deleted](/components/branches/tutorial/figures/branch-deleted.png) If you go to **Storage**, you will see that the bucket `out.c-bitcoin` still only has the `top_prices` table. The table `dollar_btc_transactions` is missing even though you had it in your branch and you merged the configuration. -![Screenshot - Production storage after branch is deleted](/getting-started/branches/figures/branch-deleted-storage.png) +![Screenshot - Production storage after branch is deleted](/components/branches/tutorial/figures/branch-deleted-storage.png) This is expected. Branch storage is completely isolated and no data are merged back to production, only configurations. You need to run the connector in production to get the data into production **Storage**. Also, notice that the branch diff --git a/src/content/docs/getting-started/branches/prepare-files.md b/src/content/docs/components/branches/tutorial/prepare-files.md similarity index 62% rename from src/content/docs/getting-started/branches/prepare-files.md rename to src/content/docs/components/branches/tutorial/prepare-files.md index 22d49b610..3a90c49c9 100644 --- a/src/content/docs/getting-started/branches/prepare-files.md +++ b/src/content/docs/components/branches/tutorial/prepare-files.md @@ -1,14 +1,15 @@ --- title: 'Prepare File-Manipulating Configurations' -slug: 'getting-started/branches/prepare-files' +slug: 'components/branches/tutorial/prepare-files' description: "Set up the file-manipulating configurations used by the development branches walkthrough: a Python transformation that writes files to Storage." redirect_from: - /tutorial/branches/prepare-files/ + - /getting-started/branches/prepare-files/ --- -In the [previous part](/getting-started/branches/prepare-tables/) of our tutorial on using development branches, you prepared +In the [previous part](/components/branches/tutorial/prepare-tables/) of our tutorial on using development branches, you prepared production configurations that manipulate tables. Now you will create the production configurations that work with [files](/storage/files/). @@ -16,7 +17,7 @@ with [files](/storage/files/). Let's create a production Python transformation with a simple code first. In your testing project, create a new **Python** transformation and name it `Sample Python transformation`. -![Screenshot - Create Transformation](/getting-started/branches/figures/01-new-transformation.png) +![Screenshot - Create Transformation](/components/branches/tutorial/figures/01-new-transformation.png) Add a new code in `Block 1` named `Hello world`, insert the following code, and save it. @@ -28,27 +29,27 @@ f.close () ``` -![Screenshot - New codeblock](/getting-started/branches/figures/python-new-codeblock.png) +![Screenshot - New codeblock](/components/branches/tutorial/figures/python-new-codeblock.png) ## Set Output Mapping Now go to the section **File Output Mapping** and click **New File Output**. Because the output of the transformation will be the file `demoFile.txt`, let’s set it as *Source* and `demoOutput` as *Tags*. This means that the output will be stored in Storage as `demoFile.txt` with the tag `demoOutput`. Click **Add File Output**. -![Screenshot - Set Output Mapping](/getting-started/branches/figures/05-output-mapping.png) +![Screenshot - Set Output Mapping](/components/branches/tutorial/figures/05-output-mapping.png) Here is the finished transformation. -![Screenshot - Python transformation overview](/getting-started/branches/figures/python-prod-overview.png) +![Screenshot - Python transformation overview](/components/branches/tutorial/figures/python-prod-overview.png) ## Run Transformation Now run the component. After the job is finished, go to **Storage -- Files**, where you can see the file `demoFile.txt` generated. -![Screenshot - Generated File](/getting-started/branches/figures/07-generated-file.png) +![Screenshot - Generated File](/components/branches/tutorial/figures/07-generated-file.png) At this point, you have everything ready. You created production configurations for both tables and files. It is time to take the next step: -- Learn how [tables](/getting-started/branches/tables-in-branch/) work in branches. -- Learn how [files](/getting-started/branches/files-in-branch/) work in branches. +- Learn how [tables](/components/branches/tutorial/tables-in-branch/) work in branches. +- Learn how [files](/components/branches/tutorial/files-in-branch/) work in branches. diff --git a/src/content/docs/getting-started/branches/prepare-tables.md b/src/content/docs/components/branches/tutorial/prepare-tables.md similarity index 67% rename from src/content/docs/getting-started/branches/prepare-tables.md rename to src/content/docs/components/branches/tutorial/prepare-tables.md index dc7b7b36c..491261d43 100644 --- a/src/content/docs/getting-started/branches/prepare-tables.md +++ b/src/content/docs/components/branches/tutorial/prepare-tables.md @@ -1,9 +1,10 @@ --- title: 'Prepare Table-Manipulating Configurations' -slug: 'getting-started/branches/prepare-tables' +slug: 'components/branches/tutorial/prepare-tables' description: "Set up the table-manipulating configurations used by the development branches walkthrough: a data source connector and an SQL transformation in production." redirect_from: - /tutorial/branches/prepare-tables/ + - /getting-started/branches/prepare-tables/ --- @@ -20,17 +21,17 @@ branches in a non-empty project. Let's start with a data pipeline that pulls data about bitcoin prices and creates a list of top five days when the price was the highest. You will prepare the production configurations, so that you can try working with branches later. -Start by pulling the bitcoin data. To simplify this, you can download a [prepared CSV file](/getting-started/branches/bitcoin_price.csv) using the [HTTP data source connector](/components/extractors/storage/http/). +Start by pulling the bitcoin data. To simplify this, you can download a [prepared CSV file](/components/branches/tutorial/bitcoin_price.csv) using the [HTTP data source connector](/components/extractors/storage/http/). ### Set Up Connector Create a new [HTTP connector](/components/extractors/storage/http/) configuration. Fill in **Base URL** to `https://help.keboola.com`. Then add a new table to the connector, named `bitcoin_price`, and fill in the **Path** -to `/getting-started/branches/bitcoin_price.csv`. **Table Name** should be `bitcoin_price`. +to `/components/branches/tutorial/bitcoin_price.csv`. **Table Name** should be `bitcoin_price`. -![Prepared HTTP extractor row](/getting-started/branches/figures/http-ex-prod-row.png) +![Prepared HTTP extractor row](/components/branches/tutorial/figures/http-ex-prod-row.png) -![Prepared HTTP extractor](/getting-started/branches/figures/http-ex-prod-set-up.png) +![Prepared HTTP extractor](/components/branches/tutorial/figures/http-ex-prod-set-up.png) Run the connector, and verify that a new table `in.c-keboola-ex-http-682373219.bitcoin_price` was created. @@ -40,15 +41,15 @@ on the screenshot.* ### Set Up Transformation Create a new [Snowflake transformation](/transformations/snowflake-plain/) named `Bitcoin`. -![New snowflake transformation](/getting-started/branches/figures/new-snflk.png) +![New snowflake transformation](/components/branches/tutorial/figures/new-snflk.png) In the **Table Input Mapping** section, fill in the table `bitcoin_price` that you created by running the HTTP connector. -![Snowflake input mapping](/getting-started/branches/figures/snflk-prod-im.png) +![Snowflake input mapping](/components/branches/tutorial/figures/snflk-prod-im.png) In **Table Output Mapping**, add a table `top_prices` that will be created in the transformation. -![Snowflake output mapping](/getting-started/branches/figures/snflk-prod-om.png) +![Snowflake output mapping](/components/branches/tutorial/figures/snflk-prod-om.png) Finally, add a new code to `Block 1` named `Top prices` with the following query: @@ -58,15 +59,15 @@ CREATE TABLE "top_prices" AS SELECT * FROM "bitcoin_price" ORDER BY PRICE DESC L ``` -![Snowflake output mapping](/getting-started/branches/figures/snflk-prod-code1.png) +![Snowflake output mapping](/components/branches/tutorial/figures/snflk-prod-code1.png) -![Finished transformation](/getting-started/branches/figures/transformation-prod-set-up.png) +![Finished transformation](/components/branches/tutorial/figures/transformation-prod-set-up.png) Save the transformation and run it. Then verify that there is a new table `out.c-bitcoin.top_prices` containing five values from the source data -- dates and amounts from when bitcoin had the most value. -![New table](/getting-started/branches/figures/snflk-new-table.png) +![New table](/components/branches/tutorial/figures/snflk-new-table.png) -Now you have the production set up. In [the next section](/getting-started/branches/prepare-files/) of our tutorial, you'll set up +Now you have the production set up. In [the next section](/components/branches/tutorial/prepare-files/) of our tutorial, you'll set up a Python transformation using file storage. After that, you can give the branches a test run. diff --git a/src/content/docs/getting-started/branches/project-diff.md b/src/content/docs/components/branches/tutorial/project-diff.md similarity index 67% rename from src/content/docs/getting-started/branches/project-diff.md rename to src/content/docs/components/branches/tutorial/project-diff.md index 9d7d4473c..d62c0409b 100644 --- a/src/content/docs/getting-started/branches/project-diff.md +++ b/src/content/docs/components/branches/tutorial/project-diff.md @@ -1,45 +1,46 @@ --- title: 'Project Diff' -slug: 'getting-started/branches/project-diff' +slug: 'components/branches/tutorial/project-diff' description: 'Review every configuration change a development branch introduces, side by side with production, before you merge it.' redirect_from: - /tutorial/branches/project-diff/ + - /getting-started/branches/project-diff/ --- -You have already learned how [files](/getting-started/branches/files-in-branch/) and [tables](/getting-started/branches/tables-in-branch/) +You have already learned how [files](/components/branches/tutorial/files-in-branch/) and [tables](/components/branches/tutorial/tables-in-branch/) behave in branches. Now it is time to complete the branch lifecycle and merge the development branch back to production. ## Show Project Diff First, let's see what the things that you changed against the production are. To do that, switch to the `Sample branch`, go to the dashboard, and click the button **Show Project Diff** on the right. -![Screenshot - Project Diff](/getting-started/branches/figures/show-project-diff.png) +![Screenshot - Project Diff](/components/branches/tutorial/figures/show-project-diff.png) To see a detailed diff of the configuration changes in `Sample Python transformation`, click the three dots on the right and then **Compare with production**. -![Screenshot - Project Diff](/getting-started/branches/figures/project-diff.png) +![Screenshot - Project Diff](/components/branches/tutorial/figures/project-diff.png) You should see a highlighted configuration diff. -![Detailed diff of configuration change](/getting-started/branches/figures/diff-config-show-changed.png) +![Detailed diff of configuration change](/components/branches/tutorial/figures/diff-config-show-changed.png) With many changes to the configuration, it might be difficult to find what has changed. It helps to see the changes in context. To do that, uncheck the checkbox **Show changed parts only**. -![Detailed diff of configuration change](/getting-started/branches/figures/diff-config-show-all.png) +![Detailed diff of configuration change](/components/branches/tutorial/figures/diff-config-show-all.png) Back on the project diff page, note the message above the list of changed configurations. It says that your production storage will not be affected by the merge. This means that no tables will be created in production unless you run the configurations that created them, and no data will be transferred from the branch to production. For example, the table `bitcoin_transactions` which you -[created in branch](/getting-started/branches/tables-in-branch/#extend-transformation) will not be transferred to production, +[created in branch](/components/branches/tutorial/tables-in-branch/#extend-transformation) will not be transferred to production, and if the branch is deleted after the merge, the branch version of the table will be discarded as well. The table `bitcoin_transactions` will be created by running the HTTP data source connector in production after you merge it. Also, if you want to drop a column from a table in the branch, you need to drop that column from the production table as well after you merge the branch. -Continue to the last part of our tutorial where we will show you how to [merge the branch back to production](/getting-started/branches/merge-to-production/). \ No newline at end of file +Continue to the last part of our tutorial where we will show you how to [merge the branch back to production](/components/branches/tutorial/merge-to-production/). \ No newline at end of file diff --git a/src/content/docs/getting-started/branches/tables-in-branch.md b/src/content/docs/components/branches/tutorial/tables-in-branch.md similarity index 70% rename from src/content/docs/getting-started/branches/tables-in-branch.md rename to src/content/docs/components/branches/tutorial/tables-in-branch.md index 88c3d8de4..86075477f 100644 --- a/src/content/docs/getting-started/branches/tables-in-branch.md +++ b/src/content/docs/components/branches/tutorial/tables-in-branch.md @@ -1,18 +1,19 @@ --- title: 'Working with Tables in a Branch' -slug: 'getting-started/branches/tables-in-branch' +slug: 'components/branches/tutorial/tables-in-branch' description: 'See how buckets and tables behave inside a development branch, and how branch table names differ from their production counterparts.' redirect_from: - /tutorial/branches/tables-in-branch/ + - /getting-started/branches/tables-in-branch/ --- -So far, you have prepared [table](/getting-started/branches/prepare-tables/) and [file](/getting-started/branches/prepare-files/) +So far, you have prepared [table](/components/branches/tutorial/prepare-tables/) and [file](/components/branches/tutorial/prepare-files/) manipulating configurations in production. In this section of our tutorial, you will manipulate the configurations, run them, and learn how tables behave in branches. -Let's say that you want to make some changes to the [previously created configurations](/getting-started/branches/prepare-tables/) to +Let's say that you want to make some changes to the [previously created configurations](/components/branches/tutorial/prepare-tables/) to - use the top ten values instead of the top five, and - convert bitcoin (BTC) to dollars based on the value of bitcoin on the given day. @@ -26,15 +27,15 @@ your changes safely before merging them into production. To create a new branch, click on your project’s name at the top of the screen. Then click on the green icon **New** displayed next to your project’s name. -![Screenshot - Create Development Branch](/getting-started/branches/figures/08-create-dev-branch.png) +![Screenshot - Create Development Branch](/components/branches/tutorial/figures/08-create-dev-branch.png) Name your new development branch `Sample branch`, and click **Create Development Branch** to open it. -![Screenshot - Name Development Branch](/getting-started/branches/figures/09-name-dev-branch.png) +![Screenshot - Name Development Branch](/components/branches/tutorial/figures/09-name-dev-branch.png) The new branch will appear right below the name of your production project. -![Screenshot - Created Development Branch](/getting-started/branches/figures/10-dev-branch-created.png) +![Screenshot - Created Development Branch](/components/branches/tutorial/figures/10-dev-branch-created.png) ## Change Transformation @@ -43,7 +44,7 @@ the top 5. In your branch, navigate to **Transformations**. You can see the prev It is, however, only a copy living in the branch. You can verify that you are working with the branch copy by checking whether the page header has yellow accent and shows the branch name. -![Screenshot - Transformation in branch](/getting-started/branches/figures/snflk-in-branch.png) +![Screenshot - Transformation in branch](/components/branches/tutorial/figures/snflk-in-branch.png) When you are in the branch, change the query limit from `LIMIT 5` to `LIMIT 10` in the transformation code, and save it. @@ -55,7 +56,7 @@ ALTER TABLE "top_prices" DROP COLUMN "_timestamp"; ``` -![Screenshot - Change to code](/getting-started/branches/figures/transformation-branch-change-top-5.png) +![Screenshot - Change to code](/components/branches/tutorial/figures/transformation-branch-change-top-5.png) If you want, check that the original transformation did not change. You can do so by switching back to the project in the top menu. The production transformation still contains `LIMIT 5`. Now that you have verified that a branch transformation @@ -64,9 +65,9 @@ can be changed independently without affecting your production transformation, s Navigate to the transformation and run it. Examine the **Mapping** section in the job detail. -![Mapping in branch](/getting-started/branches/figures/mapping-in-branch.png) +![Mapping in branch](/components/branches/tutorial/figures/mapping-in-branch.png) -![Mapping in branch](/getting-started/branches/figures/mapping-in-branch-2.png) +![Mapping in branch](/components/branches/tutorial/figures/mapping-in-branch-2.png) Notice how the **Input** section shows the data loaded from the `bitcoin_price` table, even though you did not run the data source connector in this branch. When a branch version of a table does not exist, Keboola uses the production version as a fall back. @@ -82,31 +83,31 @@ This part of your task is done. Let's get to the second part. ## Download Transactions The second objective is to produce a list of transactions showing the amount in both BTC and USD. You need to extract -the account data first. Use the [prepared CSV file](/getting-started/branches/bitcoin_transactions.csv) with sample transactions. +the account data first. Use the [prepared CSV file](/components/branches/tutorial/bitcoin_transactions.csv) with sample transactions. In the branch, navigate to your existing HTTP connector configuration and add a new table named `bitcoin_transactions`. -Fill in **Path** with `/getting-started/branches/bitcoin_transactions.csv`. Save and run the connector. +Fill in **Path** with `/components/branches/tutorial/bitcoin_transactions.csv`. Save and run the connector. -![Extractor for transactions](/getting-started/branches/figures/extractor-transactions.png) +![Extractor for transactions](/components/branches/tutorial/figures/extractor-transactions.png) -![Extractor for transactions](/getting-started/branches/figures/extractor-transactions-2.png) +![Extractor for transactions](/components/branches/tutorial/figures/extractor-transactions-2.png) Examine the job outputs. There are tables with the branch icon in a bucket prefixed with a number. As we've already shown, this means that the output was stored in branch context, keeping your production data intact. -![Extractor output](/getting-started/branches/figures/extractor-output.png) +![Extractor output](/components/branches/tutorial/figures/extractor-output.png) If you navigate to **Storage Explorer** in the branch, you'll see that the created buckets are shown there with the icon as well. -![Branch storage](/getting-started/branches/figures/dev-branch-storage.png) +![Branch storage](/components/branches/tutorial/figures/dev-branch-storage.png) Also, when you switch back to production and navigate to the **Storage Explorer**, you'll see a switch to show the branch buckets. -![Storage explorer in production](/getting-started/branches/figures/storage-dev-buckets.png) +![Storage explorer in production](/components/branches/tutorial/figures/storage-dev-buckets.png) -![Storage explorer in production](/getting-started/branches/figures/storage-dev-buckets-2.png) +![Storage explorer in production](/components/branches/tutorial/figures/storage-dev-buckets-2.png) ***Note:** There is only one [Storage](/storage/) in your project. Branch buckets are created with a prefixed name, but are stored along normal buckets and are visible in production as well as in a branch. You can see all buckets @@ -115,7 +116,7 @@ from all development branches in production.* Now switch back to the `Sample branch`. Navigate to the `Bitcoin` transformation and run it again. Check the input mapping and verify that the `bitcoin_prices` table from the branch was used as input. -![Input mapping from branch data](/getting-started/branches/figures/input-mapping-from-branch.png) +![Input mapping from branch data](/components/branches/tutorial/figures/input-mapping-from-branch.png) What happened is that Keboola checked whether a branch version of the bucket existed and since it did (you ran the branch version of the HTTP connector), it was used in the transformation. @@ -124,16 +125,16 @@ What happened is that Keboola checked whether a branch version of the bucket exi You'll be loading the data from the table `bitcoin_transactions`, so you need to add the table to input mapping. -![Input mapping from branch data](/getting-started/branches/figures/transformation-branch-input-mapping.png) +![Input mapping from branch data](/components/branches/tutorial/figures/transformation-branch-input-mapping.png) Notice the branch icon next to the table name. You're referring to the branch version of the table. Don't be alarmed by the UI saying that the table does not exist -- it exists only in the branch so far. The UI will be improved in future versions. -![Missing table in input mapping](/getting-started/branches/figures/transformation-branch-input-mapping-missing.png) +![Missing table in input mapping](/components/branches/tutorial/figures/transformation-branch-input-mapping-missing.png) Now add a second code to `Block 1` named `Dollar values of transactions` and insert the following SQL: -![Add a new code](/getting-started/branches/figures/transformation-branch-add-code.png) +![Add a new code](/components/branches/tutorial/figures/transformation-branch-add-code.png) ```sql @@ -151,18 +152,18 @@ LEFT JOIN ``` -![Add a new code](/getting-started/branches/figures/transformation-branch-added-code.png) +![Add a new code](/components/branches/tutorial/figures/transformation-branch-added-code.png) For the data to make it out of the transformation, you need to add the created `dollar_btc_transactions` table to output mapping as well. -![Output mapping in branch](/getting-started/branches/figures/output-mapping-branch-transformation.png) +![Output mapping in branch](/components/branches/tutorial/figures/output-mapping-branch-transformation.png) -![Changed transformation overview](/getting-started/branches/figures/transformation-branch-overview.png) +![Changed transformation overview](/components/branches/tutorial/figures/transformation-branch-overview.png) Run the updated transformation and examine the results. -![Changed transformation output](/getting-started/branches/figures/transformation-branch-output.png) +![Changed transformation output](/components/branches/tutorial/figures/transformation-branch-output.png) As you can see, this run of the transformation only accessed branch buckets. It loaded the data from the HTTP connector from the branch bucket as input and stored the output in two tables in another branch bucket. You can also examine @@ -170,5 +171,5 @@ the data in the tables to see that you indeed created a list of transactions wit The second part of your task is done as well. You changed the table-manipulating production configurations in a branch. You verified that the none of the changes affected the original project configurations or the project production -data. Next you will run the [file-manipulating configuration in a branch](/getting-started/branches/files-in-branch/) and +data. Next you will run the [file-manipulating configuration in a branch](/components/branches/tutorial/files-in-branch/) and examine how files work there. \ No newline at end of file diff --git a/src/content/docs/components/extractors/database/bigquery/index.md b/src/content/docs/components/extractors/database/bigquery/index.md index f229580dd..f597012fa 100644 --- a/src/content/docs/components/extractors/database/bigquery/index.md +++ b/src/content/docs/components/extractors/database/bigquery/index.md @@ -15,7 +15,7 @@ Running the connector creates a background job that - exports the results from Google Cloud Storage and stores them in specified tables in Keboola Storage. - removes the results from Google Cloud Storage. -***Note:** Using the Google BigQuery connector is also described in our [Getting Started Tutorial](/getting-started/ad-hoc/#using-bigquery-connector).* +***Note:** Using the Google BigQuery connector is also described in our [Getting Started Tutorial](/workspace/ad-hoc-analysis/#using-bigquery-connector).* ## Initial Setup diff --git a/src/content/docs/components/extractors/database/index.md b/src/content/docs/components/extractors/database/index.md index e09925cb3..5d9065b95 100644 --- a/src/content/docs/components/extractors/database/index.md +++ b/src/content/docs/components/extractors/database/index.md @@ -34,7 +34,7 @@ This straightforward approach suits most use cases and supports Timestamp-based All are [configured](/components/extractors/database/sqldb/#initial-setup) similarly and offer an [advanced mode](/components/extractors/database/sqldb/#advanced-mode). -Their basic configuration is also part of the [Tutorial - Loading Data from Database](/getting-started/load/database/). +Their basic configuration is covered in [Initial Setup](/components/extractors/database/sqldb/#initial-setup), which you can walk through against [our sample database](/components/extractors/database/sqldb/#try-it-with-our-sample-database). ### Log-Based Connectors diff --git a/src/content/docs/components/extractors/database/ms-sql/index.md b/src/content/docs/components/extractors/database/ms-sql/index.md index 7470306b3..585e03b70 100644 --- a/src/content/docs/components/extractors/database/ms-sql/index.md +++ b/src/content/docs/components/extractors/database/ms-sql/index.md @@ -16,7 +16,7 @@ It offers a straightforward approach suitable for most use cases, enabling [time All SQL database connectors are [configured](/components/extractors/database/sqldb/#initial-setup) similarly and offer an [advanced mode](/components/extractors/database/sqldb/#advanced-mode). -For guidance on basic configuration, please refer to our tutorial: [Loading Data with Database data source connector](/getting-started/load/database/). +For guidance on basic configuration, see [Initial Setup](/components/extractors/database/sqldb/#initial-setup). ## CDC (Change Data Capture) Mode diff --git a/src/content/docs/components/extractors/database/mysql/index.md b/src/content/docs/components/extractors/database/mysql/index.md index 13fd0ec11..fbff1e394 100644 --- a/src/content/docs/components/extractors/database/mysql/index.md +++ b/src/content/docs/components/extractors/database/mysql/index.md @@ -20,7 +20,7 @@ It is a straightforward approach suitable for most use cases, allowing for [time All connectors are [configured](/components/extractors/database/sqldb/#initial-setup) similarly and offer an [advanced mode](/components/extractors/database/sqldb/#advanced-mode). -Basic configuration is covered in the [Tutorial - Loading Data from Database](/getting-started/load/database/). +Basic configuration is covered in [Initial Setup](/components/extractors/database/sqldb/#initial-setup). ## MySQL Log-Based CDC diff --git a/src/content/docs/components/extractors/database/oracle/index.md b/src/content/docs/components/extractors/database/oracle/index.md index b9a5a9c95..2f2ef735d 100644 --- a/src/content/docs/components/extractors/database/oracle/index.md +++ b/src/content/docs/components/extractors/database/oracle/index.md @@ -13,6 +13,6 @@ It is the simplest approach suitable for most use cases and allows for [time-st They are all [configured](/components/extractors/database/sqldb/#initial-setup) in the same way and have an [advanced mode](/components/extractors/database/sqldb/#advanced-mode). -Their basic configuration is also part of the [Tutorial - Loading Data with Database Extractor](/getting-started/load/database/). +Their basic configuration is covered in [Initial Setup](/components/extractors/database/sqldb/#initial-setup). diff --git a/src/content/docs/components/extractors/database/postgresql/index.md b/src/content/docs/components/extractors/database/postgresql/index.md index 60578be0b..95f112f1c 100644 --- a/src/content/docs/components/extractors/database/postgresql/index.md +++ b/src/content/docs/components/extractors/database/postgresql/index.md @@ -21,7 +21,7 @@ It is the simplest approach suitable for most use cases and allows for [time-sta They are all [configured](/components/extractors/database/sqldb/#initial-setup) in the same way and have an [advanced mode](/components/extractors/database/sqldb/#advanced-mode). -Their basic configuration is also part of the [Tutorial - Loading Data with Database Extractor](/getting-started/load/database/). +Their basic configuration is covered in [Initial Setup](/components/extractors/database/sqldb/#initial-setup). ## PostgreSQL Log-Based CDC diff --git a/src/content/docs/components/extractors/database/sqldb/index.md b/src/content/docs/components/extractors/database/sqldb/index.md index 25e294af0..97422f5ea 100644 --- a/src/content/docs/components/extractors/database/sqldb/index.md +++ b/src/content/docs/components/extractors/database/sqldb/index.md @@ -3,6 +3,8 @@ title: Relational Sync Data Source Connectors for SQL Databases slug: 'components/extractors/database/sqldb' redirect_from: - /extractors/database/sqldb/ + - /tutorial/load/database/ + - /getting-started/load/database/ --- @@ -21,7 +23,7 @@ in the section [Server Specific Notes](#server-specific-notes). Before you start configuring your SQL data source connector, consider [setting up an SSH tunnel](/components/extractors/database/#connecting-to-database) to secure your connection to your internal database and avoid exposing your database server to the Internet. -***Note:** Our [tutorial](/getting-started/load/database/) also includes a quick introduction to extracting data from the Snowflake Database Server.* +***Note:** No database of your own to point this at? [Try it with our sample database](#try-it-with-our-sample-database) below.* ## Initial Setup After you [create a configuration](/components/#creating-component-configuration), the first step is to configure database credentials using the **Set Up Credentials First** button: @@ -44,6 +46,39 @@ Existing credentials can be changed using the **Database Credentials** link. ![Screenshot - Table list](/components/extractors/database/sqldb/sqldb-4.png) +### Try it with our sample database + +You do not need database credentials from anyone to walk the setup above — Keboola hosts a sample +Snowflake database for exactly this. Create a **Snowflake** data source configuration and enter: + +| Field | Value | +|---|---| +| Host Name | `kebooladev.snowflakecomputing.com` | +| Username, Password, Database, Schema | `HELP_TUTORIAL` | +| Warehouse | `DEV` | + +Click **Test Connection and Load Available Sources**, then select the `OPPORTUNITY`, `ACCOUNT` and +`USER` tables and click **Save and Run Configuration**. Running the connector creates a background +job which connects to the database, executes the queries, and stores the results as three new +tables in [Storage](/storage/). The procedure is identical for every database connector Keboola +supports, which is the reason to walk it once by hand. + +:::tip[Do it with Kai] +Database connectors are configured the same way as any other data source +([Integration Setup](/kai/use-cases/#integration-setup)). Open **Kai Agent** in the top bar and say +what you want connected, not what your credentials are: + +```text +Set up a Snowflake data source connector against our sample database and load the OPPORTUNITY, +ACCOUNT and USER tables into Storage. +``` + +**Never paste a password into the chat.** Kai prompts you for credentials through a secure form +instead — that is its documented behavior, and +[Kai's own guidance](/kai/getting-started/#tips-for-new-users) says the same. The values to type +into that form are the sample ones above. +::: + ## Modify Configuration If you want to modify the table extraction setup, click on the corresponding row. You'll get to the table detail view: diff --git a/src/content/docs/components/extractors/index.md b/src/content/docs/components/extractors/index.md index 2b6475924..2e6672482 100644 --- a/src/content/docs/components/extractors/index.md +++ b/src/content/docs/components/extractors/index.md @@ -30,8 +30,8 @@ Even though data source connectors are generally designed for [**automated and r they can be triggered manually at any time. - For manual import of ad-hoc data, see [Data Import in Storage](/storage/files/). To load your first tables with a connector, see [Get Your Data In](/getting-started/load/). -- Configure a [sample data source connector](/getting-started/load/googlesheets/) (Google Sheets). -- Configure a [database data source connector](/getting-started/load/database/); +- Configure a [sample data source connector](/components/extractors/storage/google-drive/) (Google Sheets). +- Configure a [database data source connector](/components/extractors/database/sqldb/#initial-setup); other SQL database data source connectors are configured in the exact same way. As bringing data into Keboola is the main purpose of a data source connectors, go the path of least resistance: diff --git a/src/content/docs/components/extractors/other/telemetry-data/telemetry-data.md b/src/content/docs/components/extractors/other/telemetry-data/telemetry-data.md index 9bdf11d2d..c856a752c 100644 --- a/src/content/docs/components/extractors/other/telemetry-data/telemetry-data.md +++ b/src/content/docs/components/extractors/other/telemetry-data/telemetry-data.md @@ -36,7 +36,7 @@ Use the toolbar to filter tables by mode, switch layout direction, and expand ta @@ -659,6 +659,11 @@ This table lists [security events](/management/project/tokens/#token-events), su - `auditLog.maintainers.updated` - `auditLog.mergeRequest.created` - `auditLog.mergeRequest.stateChanged` +- `auditLog.users.listed` +- `auditLog.users.organizationsListed` +- `auditLog.users.projectsListed` +- `auditLog.users.maintainersListed` +- `auditLog.users.superAdminGranted` - `auditLog.organization.adminAdded` - `auditLog.organization.adminRemoved` - `auditLog.organization.adminsInProjectsListed` @@ -720,6 +725,11 @@ This table lists [security events](/management/project/tokens/#token-events), su - `auditLog.storageBackendConnection.updated` - `auditLog.mergeRequest.created` - `auditLog.mergeRequest.stateChanged` +- `auditLog.users.listed` +- `auditLog.users.organizationsListed` +- `auditLog.users.projectsListed` +- `auditLog.users.maintainersListed` +- `auditLog.users.superAdminGranted` #### Operation parameters diff --git a/src/content/docs/components/extractors/storage/google-drive/index.md b/src/content/docs/components/extractors/storage/google-drive/index.md index 18bd6fa00..a2cd58a8c 100644 --- a/src/content/docs/components/extractors/storage/google-drive/index.md +++ b/src/content/docs/components/extractors/storage/google-drive/index.md @@ -1,40 +1,113 @@ --- -title: Google Drive Sheets +title: 'Google Sheets' slug: 'components/extractors/storage/google-drive' +description: 'Load tables from Google Sheets spreadsheets you own, using an authorized Google account.' redirect_from: - /extractors/storage/google-drive/ + - /tutorial/load/googlesheets/ + - /getting-started/load/googlesheets/ +--- + +This data source connector loads sheets from Google Sheets spreadsheets and stores them as tables +in your project's Storage. The component is `keboola.ex-google-drive`, and the UI lists it as +**Google Sheets — Data Source**. Unlike the +[HTTP connector](/components/extractors/storage/http/), which downloads from a public URL, this one +reads spreadsheets you own, so it needs an authorized Google account. + +Spreadsheets are commonly used for sharing small reference tables between organizations. The +walkthrough below uses a sample table so you can try the connector end to end: create a Google +spreadsheet from the [level.csv](/getting-started/level.csv) file, then imagine someone shared that +*level* table with you. + +:::tip[Do it with Kai — after you authorize] +Kai can create the configuration +([Integration Setup](/kai/use-cases/#integration-setup)). Open **Kai Agent** in the top bar: + +```text +Create a Google Sheets data source configuration called "[TUTORIAL] Level from Sheets". +``` + +Then do **steps 5–9 yourself** — both happen inside Google, not Keboola: the authorization consent +screen, and the Drive picker where you choose the *spreadsheet* you made above. After step 9, hand +it back: + +```text +In "[TUTORIAL] Level from Sheets", select the sheet inside that spreadsheet, run the +configuration, and tell me what table it created and how many rows it has. +``` + +**Check:** a new table with 28 rows, as in [step 12](#configuration). +Watch the wording — the *spreadsheet* is the file in Drive (step 9), the *sheet* is the tab inside +it (step 10). +::: + +## Prepare +Go to [Google Spreadsheets](https://www.google.com/sheets/about/) and start a new blank spreadsheet. Then go to +*File* – *Import* and upload the [level.csv](/getting-started/level.csv) file. + +![Google Spreadsheets Screenshot](/components/extractors/storage/google-drive/google-sheets-spreadsheet.png) + +## Configuration +1. Navigate to **Components** section in Keboola and click the **Add Component** button: + +![Data Source Overview Screenshot](/components/extractors/storage/google-drive/source-intro-0.png) + +2. Utilize the search box to locate the *Google Sheets data source connector*. Once found, click on it. + +![Data Source Overview Screenshot](/components/extractors/storage/google-drive/source-intro.png) + +3. Click **Connect To My Data**. The 'Use With Demo Data' option will extract datasets prepared by Keboola for your experimentation outside of this guide, and it can be found across all commonly used connectors. + +4. Enter a name and description and click **Create Configuration**. + +![Create Google Sheets Configuration](/components/extractors/storage/google-drive/google-sheets-create.png) + + Each Keboola component (data source, data destination, or application) can support multiple [*configurations*](/components/). + This concept enables you to, for instance, extract data from multiple Google accounts. + +5. Authorize the connector to access the spreadsheet by clicking the **Sign in with Google** button. + +![Sign in with Google](/components/extractors/storage/google-drive/sign-in-with-google.png) + +6. On the following screen, click **Allow**. + +![Access Google Account](/components/extractors/storage/google-drive/allow.png) + +7. Now you want to select the Google Drive files to import. + +![Select Google Drive Files](/components/extractors/storage/google-drive/select-files.png) + +8. In step 5, you authorized Keboola to use your account to access the Drive. In this step, you will be asked to grant access specifically to spreadsheets. +Click **'Select all'** and then proceed by clicking **'Continue'** on the following screen. + +![Get Access to Spreadsheets](/components/extractors/storage/google-drive/access-to-spreadsheets.png) + +9. Use the search box to find your **Level** spreadsheet. Select it and click the **Select** button. + +![Find Spreadsheet](/components/extractors/storage/google-drive/find-spreadsheet.png) + +10. Keboola has automatically detected all sheets from within your spreadsheet and will now allow you to select the one you want to load. +11. Select the sheet and click **Save and Run Configuration**. A job will be executed, and once completed, you will see a new table created. + +![Save and Run Configuration](/components/extractors/storage/google-drive/save-and-run.png) + +12. The Google Sheets data source automatically generates an output bucket and table. Click on the name of the output table to check its contents, +or navigate directly to the **Storage** section to explore the data. + +![Go to Storage](/components/extractors/storage/google-drive/storage.png) + +## Modify Configuration +When a sheet is added to the connector, it is displayed in the list of extracted sheets: + +![Screenshot - Sheet list](/components/extractors/storage/google-drive/google-drive-4.png) + +Configured tables are stored as [configuration rows](/components/#configuration-rows). +The list shows the name (and the link) of the imported document and sheet, and also the name of the destination +table in [Storage](/storage/). You can modify the destination table name by editing the sheet extraction. + +## Going further ---- - - - -This data source connector loads sheets from Google Drive Sheets and stores them as tables in a bucket in your -current project. - -## Configuration -[Create a new configuration](/components/#creating-component-configuration) of the **Google Drive** connector. -Then click **Authorize Account** to [authorize the configuration](/components/#authorization). - -Click **New Sheet** to configure extraction and **Select spreadsheet** to list accessible spreadsheets -in your account: - -![Screenshot - Empty configuration](/components/extractors/storage/google-drive/google-drive-1.png) - -You may be asked once again to log in. In that case, use the same account in which you authorized the first step of the setup. -Choose the document you want to import: - -![Screenshot - Select document](/components/extractors/storage/google-drive/google-drive-2.png) - -The sheets of the selected document are shown; you can select which sheets you want to import: - -![Screenshot - Select sheet](/components/extractors/storage/google-drive/google-drive-3.png) - -## Modify Configuration -When a sheet is added to the connector, it is displayed in the list of extracted sheets: - -![Screenshot - Sheet list](/components/extractors/storage/google-drive/google-drive-4.png) - -Configured tables are stored as [configuration rows](/components/#configuration-rows). -The list shows the name (and the link) of the imported document and sheet, and also the name of the destination -table in [Storage](/storage/). You can modify the destination table name by editing the sheet extraction. - +- [Google Sheets data destination connector](/components/writers/storage/google-sheets/) — writing + a Storage table back out into a spreadsheet. +- [Get Your Data In](/getting-started/load/) — the Getting Started step this walkthrough used to be + a side trip from. diff --git a/src/content/docs/components/extractors/storage/index.md b/src/content/docs/components/extractors/storage/index.md index f62c77931..f97e54eab 100644 --- a/src/content/docs/components/extractors/storage/index.md +++ b/src/content/docs/components/extractors/storage/index.md @@ -11,7 +11,7 @@ The following data source connectors allow access to data from generic storage s - [AWS S3](/components/extractors/storage/aws-s3/) — imports CSV files from multiple AWS S3 buckets into multiple tables with additional postprocessing. - [Azure Datalake Gen2](/components/extractors/storage/azure-datalake-gen2/) — imports CSV files from Azure Datalake Gen2 into multiple tables. - [FTP](/components/extractors/storage/ftp/) — imports CSV files from the FTP, FTPS, and SFTP servers. -- [Google Drive](/components/extractors/storage/google-drive/) — imports data from Google Sheets (also part of the [Tutorial](/getting-started/load/googlesheets/)). +- [Google Sheets](/components/extractors/storage/google-drive/) — imports sheets from Google Sheets spreadsheets, with a full walkthrough against a sample table. - [HTTP](/components/extractors/storage/http/) — imports CSV files stored on HTTP or HTTPS. - [Keboola Storage](/components/extractors/storage/storage-api/) — loads single or multiple tables from a Keboola project and stores them in a bucket in your current project; can be used where [Data Catalog](/catalog/) cannot. - [OneDrive Excel Sheets](/components/extractors/storage/onedrive-excel-sheets/) — imports data from OneDrive Excel sheets. diff --git a/src/content/docs/data-apps/authentication.md b/src/content/docs/data-apps/authentication.md deleted file mode 100644 index 6b132c840..000000000 --- a/src/content/docs/data-apps/authentication.md +++ /dev/null @@ -1,170 +0,0 @@ ---- -title: Authentication -slug: 'data-apps/authentication' -description: Control who can open your Keboola app, including how to set up OIDC with Auth0, Google Cloud, Microsoft Entra ID, or Okta. -redirect_from: - - /components/data-apps/authentication/ - - /data-apps/oidc/ - - /components/data-apps/oidc/ - - /data-apps/oidc/auth0/ - - /components/data-apps/oidc/auth0/ - - /data-apps/authentication/auth0/ - - /data-apps/oidc/google-cloud-platform/ - - /components/data-apps/oidc/google-cloud-platform/ - - /data-apps/authentication/google-cloud-platform/ - - /data-apps/oidc/microsoft-entra-id/ - - /components/data-apps/oidc/microsoft-entra-id/ - - /data-apps/authentication/microsoft-entra-id/ - - /data-apps/oidc/okta/ - - /components/data-apps/oidc/okta/ - - /data-apps/authentication/okta/ ---- - -Once an app is deployed, its URL is publicly available. Protect it so only the right people can open it, and choose the method that fits your audience. You set it in the app's configuration under **Authentication → Authentication Type**, which offers six options. - -![The Authentication Type dropdown in an app's configuration, listing None, Basic, OIDC, GitLab, GitHub, and JumpCloud](/data-apps/auth-options.png) - -## Authentication methods - -- **None (Public Access)** — the app is public to anyone with the URL. You can still add your own authorization inside the app; for Streamlit, use the [Streamlit authenticator](https://github.com/mkhorasani/Streamlit-Authenticator) ([example](https://github.com/KB-PS/mkt-bi-ocr/blob/master/Select_Invoices.py)). -- **Basic (Password)** — the **default** for new apps. Keboola generates a shared password; users enter it before the app opens. Once the app is deployed, the password is shown on the app's configuration page next to **Open App**, ready to copy — and when Kai builds an app, it shows the password as the last step. -- **OIDC (Custom)** — users sign in with your identity provider (Auth0, Google Cloud, Microsoft Entra ID, Okta). Recommended for anything beyond a quick share. -- **GitHub** — restrict access with GitHub OAuth by organization, team, repository, or allowed users. -- **GitLab** — restrict access with GitLab OAuth by groups, projects, or roles. -- **JumpCloud** — restrict access with JumpCloud OIDC, with optional role-based filtering. - -## OIDC (single sign-on) - -OIDC lets users log into your app through your single sign-on (SSO) provider. Keboola supports Auth0, Google Cloud, Microsoft Entra ID, and Okta. When you open an OIDC-protected app, you pick an **Authentication Provider** and sign in. - -![Select OIDC provider](/data-apps/auth-select-oidc-provider.png) - -### Set up OIDC - -The flow is the same for every provider — only the provider option, issuer URL, and a few provider quirks differ (see the table below). You must register a callback URL for **each** app; credentials can't be reused across apps. - -1. **Register the app with your identity provider.** Create an OIDC / OAuth 2.0 **web application** in the provider's console. You'll get a **Client ID** and **Client Secret**. Leave the callback URL for now — you don't have it until the Keboola app exists. -2. **Create the app in Keboola.** Open **Apps**, click **+ Create App**, and [create the app manually](/data-apps/getting-started/#create-an-app-manually) — pick a stack, name the app, and click **Create App**. It opens on its own configuration page. -3. **Set the authentication method.** On the configuration page, under **Authentication**, select **OIDC**, choose your provider option, and paste the **Client ID**, **Client Secret**, and **Issuer URL**. Click **Save**. -4. **Add the callback URL to your provider.** Register the app's [callback URL](#callback-url-format) as the authorized redirect URI in the provider's console. -5. **Deploy the app.** Set the app's **code source** — a Python/JS app runs from a connected **Git repository**; a Streamlit app can use inline **Code** or Git — then click **Deploy App** and complete the short wizard (backend size, inactivity timeout). -6. **Test.** Open the app URL — you should be redirected to your provider to sign in, then land in the app. - -### Provider settings - -| Provider | Keboola provider option | Issuer URL | Notes | -|---|---|---|---| -| **Auth0** | Generic OIDC | `https://.us.auth0.com/` | Register a regular web application. | -| **Google Cloud** | Google SSO | `https://accounts.google.com` | On the **OAuth consent screen**, add `keboola.com` under **Authorized domains**. | -| **Microsoft Entra ID** | Azure OIDC | *(from your tenant)* | Provide **Client ID**, **Client Secret**, and **Tenant ID**. To restrict by group, add a groups claim (**Manage → Token configuration → Add groups claim**; for large tenants, return only groups assigned to the app). | -| **Okta** | Generic OIDC | `https://.okta.com/oauth2/default` | Register an **OIDC – OpenID Connect** web app. | - -## GitHub authentication - -Restrict access to your app using GitHub OAuth. Users authenticate via their GitHub account, and you can optionally restrict access to specific organizations, teams, repositories, or individual users. - -### Required fields - -| Field | Description | Example | -|---|---|---| -| **Client ID** | Client ID from GitHub Developer Settings > OAuth Apps. | `Ov23liABCDEF123456` | -| **Client Secret** | Client Secret from the same GitHub OAuth App. | *(paste your GitHub secret)* | - -### Optional fields - -| Field | Description | Example | -|---|---|---| -| **GitHub URL** | Your GitHub Enterprise Server URL. Leave empty for public GitHub. | `https://github.com` | -| **Organization** | URL slug of your GitHub organization. Restricts access to organization members. | `my-company` | -| **Team** | URL slug of the team within the organization. Requires Organization to be set. | `data-engineers` | -| **Repository** | Restrict to repository collaborators. Format: `owner/repo-name`. | `my-company/analytics` | -| **Access Token** | Required for private org/team/repo restrictions. Needs `read:org` scope. Generate at GitHub > Settings > Developer Settings > Personal Access Tokens. | `ghp_...` | -| **Allowed Users** | Comma-separated GitHub usernames. If set, only these users can log in. | `jane-smith, john-doe` | - -### Setup instructions - -1. Go to your GitHub account **Settings > Developer Settings > OAuth Apps** and create a new OAuth App. -2. Set the **Authorization callback URL** to: `https://.hub./_proxy/callback` (e.g., `https://my-app-12345678.hub.north-europe.azure.keboola.com/_proxy/callback`). -3. Copy the **Client ID** and **Client Secret** from the created OAuth App. -4. In your Keboola app configuration, select **GitHub** as the authentication method. -5. Paste the **Client ID** and **Client Secret**. -6. Optionally configure organization, team, repository, or allowed users restrictions. -7. If you use organization, team, or repository restrictions with a private organization, provide an **Access Token** with `read:org` scope. -8. Save and redeploy your app. - -## GitLab authentication - -Restrict access to your app using GitLab OAuth. Users authenticate via their GitLab account, and you can optionally restrict access by groups, projects, or roles. - -### Required fields - -| Field | Description | Example | -|---|---|---| -| **Client ID** | Application ID from GitLab > Settings > Applications. | `a1b2c3d4e5f6...` | -| **Client Secret** | Application secret from the same GitLab application. | `gloas-xxxxxxxxxxxxxxxxxxxxxxxxxxxx` | -| **GitLab Instance URL** | Use `https://gitlab.com` for public GitLab, or your self-hosted URL. | `https://gitlab.com` | - -### Optional fields - -| Field | Description | Example | -|---|---|---| -| **Groups** | Only members of these groups can access the app. Use the URL path, not the display name. Separate multiple groups with commas. | `my-org/data-team` | -| **Projects** | Restrict access to members of these projects. Format: `namespace/project-slug`. | `my-org/analytics-app` | -| **Allowed Roles** | Leave empty to allow any role. Valid values: `guest`, `reporter`, `developer`, `maintainer`, `owner`. | `developer, maintainer` | - -### Setup instructions - -1. Go to your GitLab instance **Settings > Applications** and create a new application. -2. Set the **Redirect URI** to: `https://.hub./_proxy/callback` (e.g., `https://my-app-12345678.hub.north-europe.azure.keboola.com/_proxy/callback`). -3. Ensure the `openid`, `profile`, and `email` scopes are selected. If you use group or project restrictions, also select `read_api`. -4. Copy the **Application ID** and **Secret**. -5. In your Keboola app configuration, select **GitLab** as the authentication method. -6. Paste the **Client ID**, **Client Secret**, and **GitLab Instance URL**. -7. Optionally configure groups, projects, or allowed roles restrictions. -8. Save and redeploy your app. - -## JumpCloud authentication - -Restrict access to your app using JumpCloud OIDC. Users authenticate via their JumpCloud account, and you can optionally restrict access by roles. - -### Required fields - -| Field | Description | Example | -|---|---|---| -| **Client ID** | Client ID from JumpCloud Admin Console > SSO > your app. | `6507c80f5f2b490a...` | -| **Client Secret** | Client Secret from JumpCloud Admin Console > SSO > your app > SSO tab. Treat like a password. | *(paste your JumpCloud secret)* | -| **Issuer URL** | Pre-filled. For custom tenants, ask your JumpCloud admin for the correct issuer URL. | `https://oauth.id.jumpcloud.com/` | -| **Logout URL** | Pre-filled. Change only if your JumpCloud admin provides a different logout endpoint. | `https://oauth.id.jumpcloud.com/oauth2/sessions/logout` | - -### Optional fields - -| Field | Description | Example | -|---|---|---| -| **Allowed Roles** | Role values must match exactly what is set in JumpCloud's attribute mapping. Leave empty to allow any authenticated user. | `data-analyst, admin` | - -### Setup instructions - -1. In the **JumpCloud Admin Console**, go to **SSO** and create a new application (or use an existing one). -2. Configure the application as an **OIDC** application. -3. Set the **Redirect URI** to: `https://.hub./_proxy/callback` (e.g., `https://my-app-12345678.hub.north-europe.azure.keboola.com/_proxy/callback`). -4. Copy the **Client ID** and **Client Secret** from the SSO tab. -5. In your Keboola app configuration, select **JumpCloud** as the authentication method. -6. Paste the **Client ID**, **Client Secret**, **Issuer URL**, and **Logout URL**. -7. Optionally configure allowed roles to restrict access. -8. Save and redeploy your app. - -## Callback URL format - -All authentication methods that use OAuth or OIDC require a callback URL. The format is always: - -``` -https://.hub./_proxy/callback -``` - -For example: `https://my-app-12345678.hub.north-europe.azure.keboola.com/_proxy/callback` - -You can find your app's full URL after the first deployment in the app configuration. - ---- - -**Next:** [Publish and share →](/data-apps/publish-and-share/) diff --git a/src/content/docs/data-apps/authentication.mdx b/src/content/docs/data-apps/authentication.mdx new file mode 100644 index 000000000..321087119 --- /dev/null +++ b/src/content/docs/data-apps/authentication.mdx @@ -0,0 +1,311 @@ +--- +title: Authentication +slug: 'data-apps/authentication' +description: "Control who can open your Keboola app: a shared password, single sign-on with Google, Microsoft Entra ID, Okta or Auth0, or GitHub, GitLab and JumpCloud accounts." +redirect_from: + - /components/data-apps/authentication/ + - /data-apps/oidc/ + - /components/data-apps/oidc/ + - /data-apps/oidc/auth0/ + - /components/data-apps/oidc/auth0/ + - /data-apps/authentication/auth0/ + - /data-apps/oidc/google-cloud-platform/ + - /components/data-apps/oidc/google-cloud-platform/ + - /data-apps/authentication/google-cloud-platform/ + - /data-apps/oidc/microsoft-entra-id/ + - /components/data-apps/oidc/microsoft-entra-id/ + - /data-apps/authentication/microsoft-entra-id/ + - /data-apps/oidc/okta/ + - /components/data-apps/oidc/okta/ + - /data-apps/authentication/okta/ +--- + +import { Tabs, TabItem } from '@astrojs/starlight/components'; + +Once an app is deployed, its URL is publicly available. Protect it so only the right people can open it, and choose the method that fits your audience. You set it in the app's configuration under **Authentication → Authentication Type**, which offers six options. + +![The Authentication Type dropdown in an app's configuration, listing None, Basic, OIDC, GitLab, GitHub, and JumpCloud](/data-apps/auth-options.png) + +## Authentication methods + +- **None (Public Access)** — the app is public to anyone with the URL. You can still add your own authorization inside the app; for Streamlit, use the [Streamlit authenticator](https://github.com/mkhorasani/Streamlit-Authenticator) ([example](https://github.com/keboola/mkt-bi-ocr/blob/master/Select_Invoices.py)). +- **Basic (Password)** — the **default** for new apps. Keboola generates a shared password; users enter it before the app opens. Once the app is deployed, the password is shown on the app's configuration page next to **Open App**, ready to copy — and when Kai builds an app, it shows the password as the last step. +- **OIDC (Custom)** — users sign in with your identity provider (Google, Microsoft Entra ID, Okta, Auth0, or any other OIDC provider). Recommended for anything beyond a quick share. +- **GitHub** — restrict access with GitHub OAuth by organization, team, repository, or allowed users. +- **GitLab** — restrict access with GitLab OAuth by groups, projects, or roles. +- **JumpCloud** — restrict access with JumpCloud OIDC, with optional role-based filtering. + +## OIDC (single sign-on) + +OIDC lets users log into your app through your single sign-on (SSO) provider. Keboola has ready-made provider options for Google (**Google SSO**), Microsoft Entra ID (**Azure OIDC**), **Okta**, and **Auth0**, plus **Generic OIDC** for any other OpenID Connect provider. Users sign in with the provider you configured; if an app has more than one provider, they first pick an **Authentication Provider**. + +![The app's sign-in page asking the user to select an authentication provider, one button per configured provider](/data-apps/auth-select-oidc-provider.png) + +The flow is the same everywhere: create the app in Keboola, register it with your provider using the app's callback URL, paste the provider's credentials into the app's **Authentication** settings, and deploy. + +### Step 1 — Create the app and copy its callback URL + +Your provider needs the app's callback URL, so create the app first. + +1. In your Keboola project, open **Apps**, click **+ Create App**, and [create the app manually](/data-apps/getting-started/#create-an-app-manually). The app opens on its configuration page. (Adding sign-in to an existing app? Open its configuration instead.) +2. Scroll to the **App URL** block. It shows the app's host as a URL prefix plus a generated part, for example `toy-store-sales` and `-74016144.hub.europe-west3.gcp.keboola.com`. Your callback URL is `https://`, that whole host, and `/_proxy/callback`: + + ``` + https://-.hub./_proxy/callback + ``` + + For example: `https://toy-store-sales-74016144.hub.europe-west3.gcp.keboola.com/_proxy/callback` + + The block is there from the moment the app exists; you don't have to deploy first. + +![The app's configuration page with the App URL block: the URL prefix, the generated host, and a copy button](/data-apps/publish-config.png) + +Keep this tab open. Each app has its own callback URL, so register every app with your provider separately. + +### Step 2 — Set up your identity provider + +Pick your provider: + + + + Let people sign in with their Google account. + + **Before you start** — you need a Google Cloud project where you can manage the consent screen and OAuth clients: the project **Owner** role, or the **OAuth Config Editor** role. Decide who should get in, too. To limit sign-in to your Google Workspace organization, the Google Cloud project must belong to that organization and you pick the **Internal** audience below. Otherwise the audience is **External**: any Google account can sign in once you publish the app. + + **Set up the consent screen:** + + 1. Open the [Google Cloud console](https://console.cloud.google.com/), select your project, and open **Google Auth Platform** (search for it in the console's top bar). + 2. First time in this project? Click **Get started** and fill in the wizard: **App name** and **User support email**, the **Audience** (**Internal** for your Google Workspace organization only, **External** for any Google account), and a contact email, then click **Create**. + 3. Open **Branding**. Under **Authorized domains**, add `keboola.com`, the domain your App URL belongs to, and save. If Google refuses the redirect URI in the next step, it's because the domain is missing here. + 4. Chose **External**? Open **Audience** and add yourself and your testers under **Test users** (up to 100), or click **Publish app** to let anyone with a Google account sign in. Until you do one of these, Google turns everyone else away. + + **Create the OAuth client:** + + 1. Open **Clients** and click **Create client**. + 2. Set **Application type** to **Web application** and give the client a name, for example `Keboola app - Toy store sales`. + 3. Under **Authorized redirect URIs**, click **Add URI** and paste the callback URL from step 1. It has to match exactly: `https`, the full host, and `/_proxy/callback` with no trailing slash. + 4. Click **Create**. Copy the **Client ID** and the **Client secret** now; Google shows the secret only at creation. If you lose it, open the client and click **Add Secret**. + + **Back in Keboola** — on the app's configuration page, under **Authentication**, set **Authentication Type** to **OIDC (Custom)**, select **Google SSO** in the **Provider** dropdown, and paste the **Client ID** and **Client secret**. There's no issuer field for this option; Keboola uses `https://accounts.google.com`. Click **Save**. + + **If sign-in fails:** + + - `Error 400: redirect_uri_mismatch` — the URI in the OAuth client differs from the app's callback URL. Compare them character by character (scheme, host, `/_proxy/callback`, no trailing slash) and fix the client. + - "Access blocked: … has not completed the Google verification process", or a colleague can't get past Google — the audience is External and the app is still in **Testing**, so only listed **Test users** can sign in. Add them on the **Audience** page, or click **Publish app**. + - Someone outside your organization can't sign in — expected with the Internal audience. Switch to External on the **Audience** page if that's not what you want. + - `invalid_client`, or a token error right after signing in — the Client ID or Client secret in Keboola doesn't match the OAuth client. Paste them again, or add a new secret in Google Cloud and update the app. + + Changed the redirect URI or the audience on Google's side later? No redeploy needed; Google says such changes take from a few minutes to a few hours to apply. + + + Let people sign in with their Microsoft work account. + + **Before you start** — you need permission to register applications in your tenant; the **Application Developer** role is enough for the registration itself. Restricting who can sign in needs **Cloud Application Administrator** or higher. + + **Register the app:** + + 1. Sign in to the [Microsoft Entra admin center](https://entra.microsoft.com/) and go to **Entra ID → App registrations → New registration**. + 2. Enter a **Name** your users will recognize. + 3. Under **Supported account types**, keep **Single tenant only** (the option names your tenant): only users and guests of your tenant can sign in. + 4. Under **Redirect URI (optional)**, choose **Web** and paste the callback URL from step 1. + 5. Click **Register**. The app's **Overview** page opens. Copy the **Application (client) ID** and the **Directory (tenant) ID**; Keboola needs both. + + **Create a client secret:** + + 1. Under **Manage**, open **Certificates & secrets** and click **New client secret**. + 2. Enter a description, pick an expiry, and click **Add**. + 3. Copy the secret's **Value** right away; Entra hides it once you leave the page. Note the expiry, too: before it passes, create a new secret and update the app's authentication settings, or sign-in stops working. + + **Optional — restrict who can sign in.** Out of the box, anyone in your tenant can open the app. To let in only specific people or groups, go to **Entra ID → Enterprise apps**, open your app, and under **Manage → Properties** set **Assignment required?** to **Yes**. Then, under **Manage → Users and groups**, click **Add user/group** and pick who may sign in. Assigning groups, rather than single users, needs a Microsoft Entra ID P1 or P2 license. Everyone else now gets `AADSTS50105` at sign-in. + + **Back in Keboola** — on the app's configuration page, under **Authentication**, set **Authentication Type** to **OIDC (Custom)**, select **Azure OIDC** in the **Provider** dropdown, and paste the **Client ID**, **Client secret**, and **Tenant ID**. Keboola builds the issuer from the tenant ID (`https://login.microsoftonline.com//v2.0`). Click **Save**. + + **If sign-in fails:** + + - `AADSTS50011: The redirect URI … does not match` — the redirect URI in the registration differs from the app's callback URL. Fix it under **Manage → Authentication**. + - `AADSTS7000215: Invalid client secret provided` — you pasted the secret's **Secret ID** instead of its **Value**, or the secret expired. Create a new one and update the app. + - `AADSTS50105: … is not assigned to a role for the application` — you required assignment and this user isn't assigned. Add them under **Enterprise apps → your app → Users and groups**. + + + Let people sign in with their Okta account. + + **Before you start** — you need admin access to your Okta org (the **Application Administrator** or **Super Administrator** role) to create app integrations. + + **Create the app integration:** + + 1. In the Okta Admin Console, go to **Applications → Applications** (in newer orgs the menu is called **Applications and Resources**) and click **Create App Integration**. + 2. Choose **OIDC - OpenID Connect** as the **Sign-in method** and **Web Application** as the **Application type**, then click **Next**. + 3. Enter an **App integration name**, for example `Keboola app - Toy store sales`. + 4. Under **Sign-in redirect URIs**, replace the default value with the callback URL from step 1. You can remove the default **Sign-out redirect URIs** entry. + 5. Under **Assignments**, decide who can sign in: **Allow everyone in your organization to access** or **Limit access to selected groups** (or skip the assignment for now and add people later). + 6. Click **Save**. On the **General** tab, under **Client Credentials**, copy the **Client ID** and the **Client secret**. + + **Back in Keboola** — on the app's configuration page, under **Authentication**, set **Authentication Type** to **OIDC (Custom)**, select **Okta** in the **Provider** dropdown, and paste the **Client ID** and **Client secret**. Set **Domain/Org URL** to `https:///oauth2/default`; your Okta domain is shown in the Admin Console when you click your name at the top right, for example `https://acme.okta.com/oauth2/default`. The `default` authorization server comes with the Integrator Free Plan and with API Access Management; if **Security → API → Authorization Servers** doesn't list it, use the org authorization server instead: `https://` with no path. **Logout URL** is optional: `https:///login/signout` also ends the Okta session when someone signs out of the app. Click **Save**. + + **If sign-in fails:** + + - "The 'redirect_uri' parameter must be a Login redirect URI in the client app settings" — the URI in the integration differs from the app's callback URL. Fix **Sign-in redirect URIs** on the integration's **General** tab. + - "User is not assigned to the client application" — the user, or their group, isn't assigned to the integration. Add them on the **Assignments** tab. + - Issuer or discovery error when the app starts the sign-in — check **Domain/Org URL**: `https://`, your Okta domain, `/oauth2/default`, no trailing slash. If your org has no `default` authorization server, use `https://` instead. + + + Let people sign in through Auth0, with whatever connections your tenant offers: username and password, social logins, or enterprise identity providers. + + **Before you start** — you need the **Admin** role on your Auth0 tenant; the Editor roles can't create applications. + + **Register the application:** + + 1. In the [Auth0 Dashboard](https://manage.auth0.com/), go to **Applications → Applications** and click **Create Application**. + 2. Name it, choose **Regular Web Applications**, and click **Create**. + 3. Open the **Settings** tab. Under **Basic Information**, copy the **Domain**, **Client ID**, and **Client Secret**. + 4. Scroll down to **Application URIs** and paste the callback URL from step 1 into **Allowed Callback URLs**. + 5. Click **Save Changes** at the bottom of the page. + + **Back in Keboola** — on the app's configuration page, under **Authentication**, set **Authentication Type** to **OIDC (Custom)**, select **Auth0** in the **Provider** dropdown, and paste the **Client ID** and **Client secret**. Set **Issuer URL** to your tenant **Domain** with `https://` in front and a trailing slash, for example `https://acme.us.auth0.com/` (or a custom domain such as `https://login.acme.com/`). The field's example omits the slash, but Auth0's issuer ends with one; when in doubt, copy the `issuer` value from `https:///.well-known/openid-configuration`. **Logout URL** is optional: `https:///oidc/logout` also ends the Auth0 session when someone signs out of the app. Click **Save**. + + **If sign-in fails:** + + - "Callback URL mismatch" on an Auth0 error page — the URL in **Allowed Callback URLs** differs from the app's callback URL. Fix it under **Settings → Application URIs**. + - Issuer mismatch or discovery error when the app starts the sign-in — the **Issuer URL** must match the `issuer` in Auth0's discovery document character for character: `https://`, your Auth0 domain, trailing slash. + - Users of one connection can't sign in — that connection isn't enabled for this application. Turn it on under the application's **Connections** tab. + + + Any OpenID Connect provider works through the **Generic OIDC** option. + + 1. In the provider's console, register an OIDC / OAuth 2.0 **web application** and set the callback URL from step 1 as its redirect URI. + 2. Copy the **Client ID** and **Client secret**. + 3. Find the provider's **issuer URL**. It's the `issuer` value in the provider's discovery document, usually at `https:///.well-known/openid-configuration`. Copy it exactly, trailing slash included or omitted as the document has it — Keboola verifies it character for character. + 4. Back in Keboola, set **Authentication Type** to **OIDC (Custom)**, select **Generic OIDC** in the **Provider** dropdown, and paste the **Client ID**, **Client secret**, and **Issuer URL**. **Logout URL** is optional; it ends the provider's session when someone signs out of the app. Click **Save**. + + The Okta and Auth0 tabs walk through the same flow with a real console, so they make a useful template. + + + +### Step 3 — Deploy and test + +1. Set the app's code source and click **Deploy App**; the short wizard asks for the backend size and an inactivity timeout. (Details: [Create an app manually](/data-apps/getting-started/#create-an-app-manually). Just testing sign-in? A **Streamlit** app with a one-line inline script is the quickest thing to deploy.) +2. When the status turns **Active**, click **Open App**. Your provider asks you to sign in, then sends you into the app. + +Changing the authentication settings of an app that's already deployed? Click **Redeploy App** so the change takes effect. + +### Provider settings at a glance + +| Provider | Keboola provider option | What you enter besides Client ID and Client secret | +|---|---|---| +| **Google Cloud** | Google SSO | Nothing; the issuer `https://accounts.google.com` is preset | +| **Microsoft Entra ID** | Azure OIDC | **Tenant ID**; Keboola derives the issuer `https://login.microsoftonline.com//v2.0` | +| **Okta** | Okta | **Domain/Org URL**: `https:///oauth2/default` | +| **Auth0** | Auth0 | **Issuer URL**: `https:///` | +| Any other | Generic OIDC | **Issuer URL** from the provider; **Logout URL** optional | + +`` and `` are your tenant hosts, for example `acme.okta.com` or `acme.us.auth0.com`. + +## GitHub authentication + +Restrict access to your app using GitHub OAuth. Users authenticate via their GitHub account, and you can optionally restrict access to specific organizations, teams, repositories, or individual users. + +### Required fields + +| Field | Description | Example | +|---|---|---| +| **Client ID** | Client ID from GitHub Developer Settings > OAuth Apps. | `Ov23liABCDEF123456` | +| **Client Secret** | Client Secret from the same GitHub OAuth App. | *(paste your GitHub secret)* | + +### Optional fields + +| Field | Description | Example | +|---|---|---| +| **GitHub URL** | Your GitHub Enterprise Server URL. Leave empty for public GitHub. | `https://github.com` | +| **Organization** | URL slug of your GitHub organization. Restricts access to organization members. | `my-company` | +| **Team** | URL slug of the team within the organization. Requires Organization to be set. | `data-engineers` | +| **Repository** | Restrict to repository collaborators. Format: `owner/repo-name`. | `my-company/analytics` | +| **Access Token** | Required for private org/team/repo restrictions. Needs `read:org` scope. Generate at GitHub > Settings > Developer Settings > Personal Access Tokens. | `ghp_...` | +| **Allowed Users** | Comma-separated GitHub usernames. If set, only these users can log in. | `jane-smith, john-doe` | + +### Setup instructions + +1. Go to your GitHub account **Settings > Developer Settings > OAuth Apps** and create a new OAuth App. +2. Set the **Authorization callback URL** to: `https://.hub./_proxy/callback` (e.g., `https://my-app-12345678.hub.north-europe.azure.keboola.com/_proxy/callback`). +3. Copy the **Client ID** and **Client Secret** from the created OAuth App. +4. In your Keboola app configuration, select **GitHub** as the authentication method. +5. Paste the **Client ID** and **Client Secret**. +6. Optionally configure organization, team, repository, or allowed users restrictions. +7. If you use organization, team, or repository restrictions with a private organization, provide an **Access Token** with `read:org` scope. +8. Save and redeploy your app. + +## GitLab authentication + +Restrict access to your app using GitLab OAuth. Users authenticate via their GitLab account, and you can optionally restrict access by groups, projects, or roles. + +### Required fields + +| Field | Description | Example | +|---|---|---| +| **Client ID** | Application ID from GitLab > Settings > Applications. | `a1b2c3d4e5f6...` | +| **Client Secret** | Application secret from the same GitLab application. | `gloas-xxxxxxxxxxxxxxxxxxxxxxxxxxxx` | +| **GitLab Instance URL** | Use `https://gitlab.com` for public GitLab, or your self-hosted URL. | `https://gitlab.com` | + +### Optional fields + +| Field | Description | Example | +|---|---|---| +| **Groups** | Only members of these groups can access the app. Use the URL path, not the display name. Separate multiple groups with commas. | `my-org/data-team` | +| **Projects** | Restrict access to members of these projects. Format: `namespace/project-slug`. | `my-org/analytics-app` | +| **Allowed Roles** | Leave empty to allow any role. Valid values: `guest`, `reporter`, `developer`, `maintainer`, `owner`. | `developer, maintainer` | + +### Setup instructions + +1. Go to your GitLab instance **Settings > Applications** and create a new application. +2. Set the **Redirect URI** to: `https://.hub./_proxy/callback` (e.g., `https://my-app-12345678.hub.north-europe.azure.keboola.com/_proxy/callback`). +3. Ensure the `openid`, `profile`, and `email` scopes are selected. If you use group or project restrictions, also select `read_api`. +4. Copy the **Application ID** and **Secret**. +5. In your Keboola app configuration, select **GitLab** as the authentication method. +6. Paste the **Client ID**, **Client Secret**, and **GitLab Instance URL**. +7. Optionally configure groups, projects, or allowed roles restrictions. +8. Save and redeploy your app. + +## JumpCloud authentication + +Restrict access to your app using JumpCloud OIDC. Users authenticate via their JumpCloud account, and you can optionally restrict access by roles. + +### Required fields + +| Field | Description | Example | +|---|---|---| +| **Client ID** | Client ID from JumpCloud Admin Console > SSO > your app. | `6507c80f5f2b490a...` | +| **Client Secret** | Client Secret from JumpCloud Admin Console > SSO > your app > SSO tab. Treat like a password. | *(paste your JumpCloud secret)* | +| **Issuer URL** | Pre-filled. For custom tenants, ask your JumpCloud admin for the correct issuer URL. | `https://oauth.id.jumpcloud.com/` | +| **Logout URL** | Pre-filled. Change only if your JumpCloud admin provides a different logout endpoint. | `https://oauth.id.jumpcloud.com/oauth2/sessions/logout` | + +### Optional fields + +| Field | Description | Example | +|---|---|---| +| **Allowed Roles** | Role values must match exactly what is set in JumpCloud's attribute mapping. Leave empty to allow any authenticated user. | `data-analyst, admin` | + +### Setup instructions + +1. In the **JumpCloud Admin Console**, go to **SSO** and create a new application (or use an existing one). +2. Configure the application as an **OIDC** application. +3. Set the **Redirect URI** to: `https://.hub./_proxy/callback` (e.g., `https://my-app-12345678.hub.north-europe.azure.keboola.com/_proxy/callback`). +4. Copy the **Client ID** and **Client Secret** from the SSO tab. +5. In your Keboola app configuration, select **JumpCloud** as the authentication method. +6. Paste the **Client ID**, **Client Secret**, **Issuer URL**, and **Logout URL**. +7. Optionally configure allowed roles to restrict access. +8. Save and redeploy your app. + +## Callback URL format + +All authentication methods that use OAuth or OIDC require a callback URL. The format is always: + +``` +https://.hub./_proxy/callback +``` + +For example: `https://my-app-12345678.hub.north-europe.azure.keboola.com/_proxy/callback` + +`` stands for the whole host shown in the **App URL** block on the app's configuration page: the URL prefix, a hyphen, and the App ID (for example `toy-store-sales-74016144`). Take that host, add `https://` in front and `/_proxy/callback` at the end. + +--- + +**Next:** [Publish and share →](/data-apps/publish-and-share/) diff --git a/src/content/docs/getting-started/branches/index.md b/src/content/docs/getting-started/branches/index.md deleted file mode 100644 index 88cecadd2..000000000 --- a/src/content/docs/getting-started/branches/index.md +++ /dev/null @@ -1,32 +0,0 @@ ---- -title: 'Development Branches' -slug: 'getting-started/branches' -description: "Change a running project safely: work in a development branch, see the project diff, and merge your changes back into production." -redirect_from: - - /tutorial/branches/ ---- - - - -The feature Development Branches allows you to modify [component configurations](/components/) -without interfering with the running configurations or entire [automated pipelines](/flows/). - -## Tutorial -First, learn how development branches [work in general](/components/branches/). - -In this tutorial, we will guide you through the process of creating and using a development branch. You will configure -various components that demonstrate the different aspects of branches. - -* Part 1 -- Preparing production configurations: - * [Preparing table manipulating configurations](/getting-started/branches/prepare-tables/) - * [Preparing file manipulating configurations](/getting-started/branches/prepare-files/) -* Part 2 -- Working in a branch: - * [Working with tables in a branch](/getting-started/branches/tables-in-branch) - * [Working with files in a branch](/getting-started/branches/files-in-branch) -* Part 3 -- Merging branches: - * [Project diff](/getting-started/branches/project-diff/) - * [Merge to production](/getting-started/branches/merge-to-production/) - -:::caution[Public Beta] -This feature is currently in public beta. Please provide feedback using the feedback button in your project. -::: diff --git a/src/content/docs/getting-started/index.md b/src/content/docs/getting-started/index.md index b816068c0..1a2a91c0d 100644 --- a/src/content/docs/getting-started/index.md +++ b/src/content/docs/getting-started/index.md @@ -127,14 +127,14 @@ assumes so you can also land on one directly and catch up. Optional side trips, once the main path makes sense. None of them are needed to finish the arc: -- **[Load from Google Sheets](/getting-started/load/googlesheets/)** and - **[Load from a Database](/getting-started/load/database/)** — load from a source that needs - credentials, rather than the public URL step 2 uses. -- **[Use a Workspace](/getting-started/transform/workspace/)** — develop and test SQL against +- **[Google Sheets](/components/extractors/storage/google-drive/)** and + **[a database](/components/extractors/database/sqldb/#try-it-with-our-sample-database)** — load + from a source that needs credentials, rather than the public URL step 2 uses. +- **[Use a Workspace](/workspace/create/)** — develop and test SQL against a copy of your data before committing it to a transformation. -- **[Ad-Hoc Data Analysis](/getting-started/ad-hoc/)** — explore arbitrary data in a Python +- **[Ad-Hoc Data Analysis](/workspace/ad-hoc-analysis/)** — explore arbitrary data in a Python or R notebook rather than building a pipeline. -- **[Development Branches](/getting-started/branches/)** — change a running project safely, +- **[Development Branches](/components/branches/tutorial/)** — change a running project safely, review the diff, then merge. ## If you are planning a rollout, not learning the tool diff --git a/src/content/docs/getting-started/load/database.md b/src/content/docs/getting-started/load/database.md deleted file mode 100644 index ea93aa2e6..000000000 --- a/src/content/docs/getting-started/load/database.md +++ /dev/null @@ -1,82 +0,0 @@ ---- -title: 'Load from a Database' -slug: 'getting-started/load/database' -description: 'Load tables from an external database using the Snowflake data source connector — the same pattern applies to every database connector Keboola supports.' -redirect_from: - - /tutorial/load/database/ ---- - -So far, you have learned to load data into Keboola [from a URL](/getting-started/load/) and -via a [Google Sheets data source connector](/getting-started/load/googlesheets/). - -Now, let's explore loading data from an external database using the Snowflake Database data source (the procedure is the same for all our database data sources). -We will use our own sample Snowflake database, so do not worry about having to get database credentials from anyone. - -:::tip[Do it with Kai] -Database connectors are configured the same way as any other data source -([Integration Setup](/kai/use-cases/#integration-setup)). Open **Kai Agent** in the top bar and -say what you want connected, not what your credentials are: - -```text -Set up a Snowflake data source connector against our sample database and load the OPPORTUNITY, -ACCOUNT and USER tables into Storage. -``` - -**Never paste a password into the chat.** Kai prompts you for credentials through a secure form -instead — that is its documented behavior, and -[Kai's own guidance](/kai/getting-started/#tips-for-new-users) says the same. The values to type into -that form are the sample ones in [step 5](#configure-snowflake-data-source-connector) below. - -The pattern is identical for every database connector Keboola supports, which is the reason to -walk it once by hand. -::: - -## Configure Snowflake Data Source Connector -1. Start by going into the **Components** section and click **Add Component**. - -![Add Data Source](/getting-started/load/db-picture1.png) - -2. Use the search box to find the **Snowflake data source**. - -![Find Snowflake Data Source](/getting-started/load/db-picture2.png) - -3. Click **Add Component** and select **Connect To My Data**. - -![Connect to Data](/getting-started/load/db-picture3.png) - -4. Enter a name and a description and click **Create Configuration**. - -![Create New Configuration](/getting-started/load/db-picture4.png) - - Similarly to other components, the Snowflake data source connector can have multiple configurations. - As each configuration represents a single database connection, we only need one configuration. - -5. Enter the following credentials: - - **Host Name** to `kebooladev.snowflakecomputing.com`. - - **Username**, **Password**, **Database**, and **Schema** to `HELP_TUTORIAL`. - - **Warehouse** to `DEV`. - -6. Click **Test Connection and Load Available Sources**. - -![Database Data Source Credentials](/getting-started/load/db-picture5.png) - -7. Under **Select sources**, use the dropdown menu to select the `OPPORTUNITY`, `ACCOUNT`, and `USER` tables. - -![Select Sources](/getting-started/load/db-picture6.png) - -8. After selecting all the required tables, click **Save and Run Configuration**. -This action will execute the data extraction, generating three new tables in your Storage. - -![Database Tables Selected](/getting-started/load/db-picture7.png) - - Running the component creates a background job that - - connects to the database, - - executes the queries, and - - stores results in the specified tables in Storage. - -For more advanced configuration options, such as incremental fetch, incremental load, or advanced SQL query mode, -please navigate to Advanced Mode. Note that we do not cover the advanced mode options here. - -![Advanced Mode](/getting-started/load/db-picture8.png) - -**Next:** [Transform your data →](/getting-started/transform/) diff --git a/src/content/docs/getting-started/load/googlesheets.md b/src/content/docs/getting-started/load/googlesheets.md deleted file mode 100644 index 133bb6b2b..000000000 --- a/src/content/docs/getting-started/load/googlesheets.md +++ /dev/null @@ -1,99 +0,0 @@ ---- -title: 'Load from Google Sheets' -slug: 'getting-started/load/googlesheets' -description: 'Load a table from an external spreadsheet using the Google Sheets data source connector, with an authorized account rather than a public URL.' -redirect_from: - - /tutorial/load/googlesheets/ ---- - -A side trip from [Get Your Data In](/getting-started/load/): the same kind of load, but from a spreadsheet you own — so the connector needs an authorized Google account instead of a public URL. - - - -Google Drive is commonly used for sharing small reference tables between different organizations. -For our purposes, create a Google spreadsheet from the [level.csv](/getting-started/level.csv) file. -Imagine someone shared the *level* table with you through Google Drive. - -:::tip[Do it with Kai — after you authorize] -Kai can create the configuration -([Integration Setup](/kai/use-cases/#integration-setup)). Open **Kai Agent** in the top bar: - -```text -Create a Google Sheets data source configuration called "[TUTORIAL] Level from Sheets". -``` - -Then do **steps 5–9 yourself** — both happen inside Google, not Keboola: the authorization consent -screen, and the Drive picker where you choose the *spreadsheet* you made above. After step 9, hand -it back: - -```text -In "[TUTORIAL] Level from Sheets", select the sheet inside that spreadsheet, run the -configuration, and tell me what table it created and how many rows it has. -``` - -**Check:** a new table with 28 rows, as in [step 12](#configure-google-sheets-data-source-connector). -Watch the wording — the *spreadsheet* is the file in Drive (step 9), the *sheet* is the tab inside -it (step 10). -::: - -## Prepare -Go to [Google Spreadsheets](https://www.google.com/sheets/about/) and start a new blank spreadsheet. Then go to -*File* – *Import* and upload the [level.csv](/getting-started/level.csv) file. - -![Google Spreadsheets Screenshot](/getting-started/load/google-sheets-spreadsheet.png) - -## Configure Google Sheets Data Source Connector -1. Navigate to **Components** section in Keboola and click the **Add Component** button: - -![Data Source Overview Screenshot](/getting-started/load/source-intro-0.png) - -2. Utilize the search box to locate the *Google Sheets data source connector*. Once found, click on it. - -![Data Source Overview Screenshot](/getting-started/load/source-intro.png) - -3. Click **Connect To My Data**. The 'Use With Demo Data' option will extract datasets prepared by Keboola for your experimentation outside of this guide, and it can be found across all commonly used connectors. - -4. Enter a name and description and click **Create Configuration**. - -![Create Google Sheets Configuration](/getting-started/load/google-sheets-create.png) - - Each Keboola component (data source, data destination, or application) can support multiple [*configurations*](/components/). - This concept enables you to, for instance, extract data from multiple Google accounts. - -5. Authorize the connector to access the spreadsheet by clicking the **Sign in with Google** button. - -![Sign in with Google](/getting-started/load/sign-in-with-google.png) - -6. On the following screen, click **Allow**. - -![Access Google Account](/getting-started/load/allow.png) - -7. Now you want to select the Google Drive files to import. - -![Select Google Drive Files](/getting-started/load/select-files.png) - -8. In step 5, you authorized Keboola to use your account to access the Drive. In this step, you will be asked to grant access specifically to spreadsheets. -Click **'Select all'** and then proceed by clicking **'Continue'** on the following screen. - -![Get Access to Spreadsheets](/getting-started/load/access-to-spreadsheets.png) - -9. Use the search box to find your **Level** spreadsheet. Select it and click the **Select** button. - -![Find Spreadsheet](/getting-started/load/find-spreadsheet.png) - -10. Keboola has automatically detected all sheets from within your spreadsheet and will now allow you to select the one you want to load. -11. Select the sheet and click **Save and Run Configuration**. A job will be executed, and once completed, you will see a new table created. - -![Save and Run Configuration](/getting-started/load/save-and-run.png) - -12. The Google Sheets data source automatically generates an output bucket and table. Click on the name of the output table to check its contents, -or navigate directly to the **Storage** section to explore the data. - -![Go to Storage](/getting-started/load/storage.png) - -## Going further - -Another side trip: load the same kind of data with the -[database data source connector](/getting-started/load/database/). - -**Next:** [Transform your data →](/getting-started/transform/) diff --git a/src/content/docs/getting-started/load/index.mdx b/src/content/docs/getting-started/load/index.mdx index 798e7fd99..f653a0b29 100644 --- a/src/content/docs/getting-started/load/index.mdx +++ b/src/content/docs/getting-started/load/index.mdx @@ -231,9 +231,9 @@ a SaaS account, and it keeps changing, so the run you kicked off by hand becomes repeats on a schedule. Both side trips are this same step against a source you have to authorize first. -- [Load from Google Sheets](/getting-started/load/googlesheets/) — a small reference table pulled +- [Google Sheets](/components/extractors/storage/google-drive/) — a small reference table pulled from a spreadsheet you own, with an authorized account instead of a public URL. -- [Load from a database](/getting-started/load/database/) — the pattern every database connector - follows. +- [A database](/components/extractors/database/sqldb/#try-it-with-our-sample-database) — the pattern + every database connector follows, walked against Keboola's own sample Snowflake database. **Next:** [Transform your data →](/getting-started/transform/) diff --git a/src/content/docs/getting-started/next-steps/index.md b/src/content/docs/getting-started/next-steps/index.md index 5b60c069c..8a6da6468 100644 --- a/src/content/docs/getting-started/next-steps/index.md +++ b/src/content/docs/getting-started/next-steps/index.md @@ -52,24 +52,24 @@ Jordan on the 08-21 call ("technically all of this could be one prompt"). --> [data source connectors](/components/extractors/) — databases, APIs, cloud storage, ad platforms, CRMs. They configure the same way the HTTP connector did, and drop into a flow the same way. Two worked examples are in this guide already: -[Google Sheets](/getting-started/load/googlesheets/) and -[a database](/getting-started/load/database/). +[Google Sheets](/components/extractors/storage/google-drive/) and +[a database](/components/extractors/database/sqldb/#try-it-with-our-sample-database). **"My transformation needs to be more than one query."** [Transformations](/transformations/) covers SQL, Python, R and dbt, code blocks and phases, shared code, and [variables](/transformations/variables/). To develop against a copy of your data -interactively, use a [workspace](/getting-started/transform/workspace/). +interactively, use a [workspace](/workspace/create/). **"I need to send data somewhere specific."** The [data destination connectors](/components/writers/) cover databases, BI tools, and storage — the Google Sheets one you used is the simplest of the family. **"I do not want to break production while I experiment."** -[Development branches](/getting-started/branches/) let you change configurations, run them, +[Development branches](/components/branches/tutorial/) let you change configurations, run them, and review a diff before merging anything into production. **"I want to explore data rather than build a pipeline."** Do -[ad-hoc analysis](/getting-started/ad-hoc/) in a Python or R workspace, or query Storage +[ad-hoc analysis](/workspace/ad-hoc-analysis/) in a Python or R workspace, or query Storage directly from a [SQL workspace](/workspace/). **"Other people need this data."** Publish it to the [Data Catalog](/catalog/) so other diff --git a/src/content/docs/getting-started/project/index.md b/src/content/docs/getting-started/project/index.md index 597827323..584d0f7f8 100644 --- a/src/content/docs/getting-started/project/index.md +++ b/src/content/docs/getting-started/project/index.md @@ -111,7 +111,7 @@ Two things. The second one is worth settling now rather than halfway through ste Not having Kai does not block the build: every step's clicking path is written out in full. The one place you will miss it is step 5's closing question, which only Kai answers — you can - run the same aggregate in a [workspace](/getting-started/transform/workspace/) instead. + run the same aggregate in a [workspace](/workspace/create/) instead. ## If it goes wrong diff --git a/src/content/docs/getting-started/transform/index.mdx b/src/content/docs/getting-started/transform/index.mdx index 5b3213fac..bb5850e9c 100644 --- a/src/content/docs/getting-started/transform/index.mdx +++ b/src/content/docs/getting-started/transform/index.mdx @@ -386,14 +386,14 @@ experimenting. [the `_timestamp` system column](/transformations/mappings/#_timestamp-system-column). The run behind these screenshots was copy-staged, so the column never appeared. - **You want to see what the query actually returns before saving.** That is what a - [workspace](/getting-started/transform/workspace/) is for. + [workspace](/workspace/create/) is for. - **You would rather not read the log yourself.** Kai has it open already ([Troubleshooting](/kai/use-cases/#troubleshooting)): `My transformation "Octopus atlas" failed. Read the last job's log and tell me what to fix.` ## Going further -- [Use a Workspace](/getting-started/transform/workspace/) — develop and test queries +- [Use a Workspace](/workspace/create/) — develop and test queries against a copy of the data before committing them to a transformation. This is how the work is really done. diff --git a/src/content/docs/getting-started/transform/workspace.md b/src/content/docs/getting-started/transform/workspace.md deleted file mode 100644 index 404c4c820..000000000 --- a/src/content/docs/getting-started/transform/workspace.md +++ /dev/null @@ -1,54 +0,0 @@ ---- -title: 'Use a Workspace' -slug: 'getting-started/transform/workspace' -description: 'Create a workspace — an isolated environment with a copy of your production data — to develop and test transformation code safely.' -redirect_from: - - /tutorial/manipulate/workspace/ ---- - -An integral aspect of creating a transformation is the development of the script itself. -In Keboola, you can use SQL, Python, or R by default. To simplify the process of writing these scripts, -we offer **workspaces** (see the full documentation of [workspaces](/workspace/)). -Workspaces provide a secure development and analytical environment -where you can interact with the data and develop your scripts with confidence. - -1. Navigate to **Workspaces** and click the **Create Workspace** button. - -![Create Workspace](/getting-started/transform/workspaces1.png) - -2. Select Snowflake SQL Workspace (or other SQL workspace depending on your project’s backend) - -![Select a Workspace](/getting-started/transform/workspaces2.png) - -3. Enter a *Name* and a *Description*. Additionally, take note that you can grant access to the workspace, allowing other users to collaborate with you. -Click **Create Workspace**. - -![Name the Workspace](/getting-started/transform/workspaces3.png) - -4. A creation job will initiate, and your workspace will soon appear among the configurations. - -![Creating Job](/getting-started/transform/workspaces4.png) - -5. Click the workspace name to access the details. - -![Access Details of the Job](/getting-started/transform/workspaces5.png) - - At the outset, you'll need to configure the **table input mapping**, much like we did when setting up a transformation. - Subsequently, click **Load Data** to clone the datasets from Storage to your workspace. The data will be cloned to your workspace - in a state as of the moment of loading. To refresh the data in a Workspace, you need to click **Load Data** again. - - If you wish to have read access to all data in your Storage without physically cloning it into the workspace, - check the *Grant read-only access to all storage data* option when creating a workspace. However, this is a feature we do not cover here. - -![Set Input Mapping](/getting-started/transform/workspaces6.png) - -6. Click **Connect**. You'll see the credentials you can use to connect to the workspace using any of your preferred IDEs. Alternatively, -click the **Connect** button again to access the Web-based Snowflake SQL IDE (please note that this only applies if your project uses a Snowflake backend). - -![Connect](/getting-started/transform/workspaces7.png) - - -After completing the development of your queries, you can then copy and paste them into a transformation configuration, -as we did in the [previous step](/getting-started/transform/). - -**Next:** [Send your data somewhere →](/getting-started/write/) diff --git a/src/content/docs/kai/best-practices.md b/src/content/docs/kai/best-practices.md index 0885f1ab7..06adcaaf9 100644 --- a/src/content/docs/kai/best-practices.md +++ b/src/content/docs/kai/best-practices.md @@ -51,6 +51,12 @@ Test changes safely by creating a development branch in the Keboola UI, then wor - **One topic per chat** — don't mix unrelated tasks - **Reset when stuck** — if Kai gets confused after 2-3 tries, start fresh with clearer context - **Let Kai read logs** — instead of pasting, use `"Read the latest job log for ex-google-analytics"` +- **Attach what Kai cannot reach** — for security reasons Kai cannot open links, so upload + a file rather than pasting a URL +- **Compact rather than lose your place** — in a long session that is still on topic, run + [`/compact`](/kai/getting-started/#compacting-a-long-conversation) and name what has to survive + the summary. It keeps the thread going where a new chat would drop everything. If the topic has + actually changed, start a new chat instead — compacting carries context you no longer want ## Security @@ -71,6 +77,9 @@ Test changes safely by creating a development branch in the Keboola UI, then wor - **Don't argue with a confused Kai** — reset the conversation instead - **Don't skip verification** — always review business logic, data quality rules, and production deployments - **Don't expect cross-project knowledge** — Kai only sees your current project +- **Don't paste a link and expect Kai to read it** — for security reasons Kai cannot open + links. Attach the document to your message, or add it as a + [context file](/kai/settings/#context-files) if you need it in every conversation ## Team Tips diff --git a/src/content/docs/kai/getting-started.md b/src/content/docs/kai/getting-started.md index 65e7914f9..0a598c43a 100644 --- a/src/content/docs/kai/getting-started.md +++ b/src/content/docs/kai/getting-started.md @@ -13,9 +13,9 @@ Kai is now in **Public Beta** and available to all users in supported stacks. Every user can see the Kai button in their project (on supported stacks). To enable Kai: -- **Organization Admins** — Can enable the feature directly from the chat screen when first clicking the Kai button -- **Other users** — Need to ask their Organization Admin to enable the feature, or [contact Keboola Support](mailto:support@keboola.com) for assistance -- **Settings** — Kai can also be enabled via **Settings → Features** in your project +- **Organization Admins** can enable the feature directly from the chat screen when first clicking the Kai button. +- **Other users** need to ask their Organization Admin to enable the feature, or [contact Keboola Support](mailto:support@keboola.com) for assistance. +- Kai can also be enabled via **Settings → Features** in your project. ## Opening Kai @@ -24,55 +24,205 @@ Click the **Kai Agent** button in your project's top bar, or use keyboard shortc | Shortcut | Action | |----------|--------| | **A** | Open the chat window (shows recent conversation) | -| **Ctrl + Shift + A** | Open a new chat | +| **Shift + A** | Open a new chat | -![Kai Chat Panel](/kai/kai-welcome.png) +![Kai open in a project](/kai/kai-open.png) ## Example Prompts -**Explore your project:** -- "What tables do we have in this project?" -- "Show me the latest job runs and their status" -- "What extractors are configured?" +**Get oriented in an unfamiliar project:** +- "What is the purpose of this project? Summarize what it does end to end." +- "What data is being ingested, from which sources, and how often?" +- "What output tables does this project produce, and what feeds each one?" +- "Trace the lineage from the raw source tables to the final output tables." + +**Understand a specific part of it:** +- "Explain what this transformation does and why the joins are shaped this way." +- "Which configurations write to the `orders` table, and which read from it?" +- "Is anything here unused? Tables nothing reads, configurations nothing runs." + +**Debug a failure:** +- "Analyze the latest failed job and tell me what went wrong." +- "This flow started failing last week. What changed in its configuration?" +- "Why did this job take 40 minutes when it usually takes 5?" + +**Maintain and improve:** +- "Help me optimize the core pipeline to reduce costs." +- "Which jobs in this project run longest, and what would you change first?" +- "This transformation full-loads every run. Can it be incremental?" + +**Build something new:** +- "Set up a Google Sheets extractor for this spreadsheet and load it into a new bucket." +- "Create a SQL transformation that calculates monthly revenue per customer from `orders`." +- "Build an integration with the Acme API that pulls orders. Its API documentation is + attached. Handle authentication and pagination." + +Building an integration for an API that has no ready-made connector is one of the stronger +things to hand Kai. For security reasons Kai cannot open links, so give it the API +documentation as a file: +attach it with [Upload files](#below-the-message-box), or add it as a +[context file](/kai/settings/#context-files) if you will be working with that API +repeatedly. A URL on its own is not enough unless the API is well known. + +For prompts that get better answers, see +[Effective Prompting](/kai/best-practices/#effective-prompting) in Best Practices. -**Analyze data:** -- "Show me the schema for the orders table" -- "How many rows are in the customers table?" +## Action Approval -**Debug issues:** -- "Analyze the latest failed job and tell me what went wrong" +Before Kai changes anything, it requests your approval: creating a configuration, +modifying a transformation, or running a job. Read-only operations do not require +approval. -**Build things:** -- "Help me create a Google Sheets extractor" -- "Create a SQL transformation that calculates monthly totals from my sales data" +Each request shows the exact parameters Kai will use. -## Action Approval +![Kai asking for approval before running a job](/kai/kai-action-approval.png) -When Kai wants to modify your project, you'll see a tool approval prompt. Review what Kai wants to do before approving. All actions are logged in your project's audit trail. +You have three options: -You can also click **Always allow** directly in the approval dialog to skip future confirmations for that specific tool. +- **Approve** — run this action once. +- **Decline** — do not run it. +- **Always allow** — run it, and stop asking for that tool in future. -For more granular control, see [Tool Permissions](/kai/settings/#tool-permissions) in Kai Settings. +All actions are logged in your project's audit trail. -## Contextual Awareness & Follow Mode +For more granular control, see [Tool Permissions](/kai/settings/#tool-permissions) in +Kai Settings. -Kai is aware of what you're currently viewing in the Keboola UI. Every message you send includes your current page location, so Kai understands your context without needing explicit references. +## Plan Mode -### How Context Works +For anything bigger than a single change — setting up a pipeline, restructuring a set of +transformations — start in plan mode. Kai explores your project read-only, then shows you what it +intends to do and waits. Nothing changes until you approve. -- **Automatic context capture** — When you send a message, Kai receives your current URL path (e.g., which configuration, job, or table you're viewing) -- **Context-aware responses** — Kai uses this information to provide relevant suggestions and can reference "this configuration" or "the current job" naturally -- **Dynamic updates** — Kai checks for the latest context during conversations, so you can navigate to different pages and Kai will adapt +Turn it on with the **Plan mode** button below the message box, or type `/plan` in your message. +They do the same thing; the button just inserts the command for you. `/plan` works anywhere in a +sentence, so "help me /plan a revenue model" is a valid plan-mode prompt. + +When Kai finishes exploring, it presents the plan as a card with three choices: + +- **Approve** — Kai leaves plan mode and starts working. Shortcut: **Cmd/Ctrl + Enter**. +- **Request changes** — say what is wrong and Kai revises the plan, staying in plan mode. +- **Dismiss** — close the plan and do nothing. + +Prefer **Request changes** over dismissing. Kai keeps everything it learned while exploring, so +revising a plan costs far less than starting over. + +The message box is hidden while a plan card is waiting, so resolve the card before you carry on +chatting. Resolving it also switches plan mode back off for you, unless you requested changes — in +that case it stays on, because Kai is still planning. + +## When Kai Asks You a Question + +When Kai needs a decision from you — which tables to model, which of two approaches to take — it +asks with clickable options instead of a paragraph of prose. Pick one and Kai carries on. + +![Kai asking which of three approaches to take](/kai/kai-clarifying-question.png) + +- Some questions take **more than one answer**; select as many as apply. +- Every question has a free-text **Other** field, so you are never limited to the options offered. +- Questions can arrive as a short series, one step at a time. The counter in the corner shows how + far through you are, and you can step back to change an earlier answer. +- **Skip** a question and Kai decides for you. + +## Chat Controls + +The chat panel has controls in two places: along the top of the panel, and below the +message box. + +### Panel header + +| Button | Control | What it does | +|--------|---------|--------------| +| ![New chat](/kai/kai-new-chat.png) | **New chat** | Start a fresh conversation. Kai keeps no context from the previous one. | +| ![Report a bug](/kai/kai-report-bug.png) | **Report a bug** | **Send support ticket** opens the support form with the details Keboola support needs — conversation ID, trace link, project, stack — pre-filled in its description. You still write the summary, pick a severity, and send it. **Copy debug info** puts the same details on your clipboard instead. On stacks without support tickets, **Report a bug** only copies. | +| ![Settings](/kai/kai-settings-gear.png) | **Settings** | Open your [Tool Permissions and System Instructions](/kai/settings/). These are personal to you and apply to this project only. Project-wide settings live in **Settings → Kai Agent**. | +| ![Expand](/kai/kai-expand-chat.png) | **Expand** | Widen the panel. The expanded view also lists your chat history, so you can reopen a previous conversation. Useful when Kai returns a long table or diagram. | +| ![Close](/kai/kai-close-chat.png) | **Close** | Close the panel. Your conversation is kept. | + +### Below the message box + +| Button | Control | What it does | +|--------|---------|--------------| +| ![Upload file](/kai/kai-upload-file.png) | **Upload files** | Attach a file or image to your message — a screenshot of an error, a sample CSV, a spec, or documentation Kai has no other way to read. See [Attaching files](#attaching-files). | +| ![Plan mode](/kai/kai-plan-mode.png) | **Plan mode** | Kai explores your project read-only, drafts a plan, and waits for your approval before changing anything. See [Plan Mode](#plan-mode). | +| ![Follow mode](/kai/kai-follow-mode.png) | **Follow mode** | Your browser navigates along as Kai works, so you can watch what it reads and modifies. Toggle it on or off at any time. | + +### Attaching files -### Follow Mode +Use **Upload files**, or drag and drop or paste straight into the chat. You can attach several +files at once. What happens next depends on the file type. -Follow mode lets you watch Kai work in real-time: +**CSV, TSV, and `.gz` files become Storage tables.** Kai opens the table-creation dialog and loads +the file into the `in.c-uploads-from-Kai` bucket, creating that bucket the first time. Kai then +works with the table, so the data is queryable like anything else in your project and outlives the +conversation. -- **Automatic navigation** — When Kai accesses a configuration, table, or other resource, your browser navigates to that page automatically -- **Visual feedback** — See exactly what Kai is reading or modifying as it happens -- **Toggle control** — Enable or disable follow mode from the chat interface based on your preference +**Every other file is attached to the conversation.** It is uploaded to your project's +[File Storage](/storage/files/) and restored each time you return to that chat, so you can refer +back to something you attached much earlier in the same conversation. Images and PDFs are read +directly by Kai, so a screenshot of a failing job or a PDF spec works as well as plain text. + +Since Kai cannot open links, a file is how you hand it anything that lives on the web. + +Two limits are worth knowing: + +- **10 MB per file.** Anything larger is skipped. +- Attachments are stored as non-permanent files, so they are **deleted after 15 days**, like any + other non-permanent file in Storage. Reopen an older chat and Kai no longer has them. + +:::tip[Do you want the file in every conversation?] +An attachment belongs to one chat and is gone after 15 days. To give Kai a document it should read +every time, such as an API reference or your naming conventions, add it as a +[context file](/kai/settings/#context-files) instead. Context files are stored permanently and Kai +reads them at the start of every conversation, until you remove them. +::: + +## Slash Commands + +Type `/` in the message box to open a searchable menu. It lists the three built-in commands below +plus any [skill files](/kai/settings/#skill-files) uploaded to your project, each with a +description, so you can find what is available without memorizing names. + +| Command | What it does | +|---------|--------------| +| `/plan` | Turn on [plan mode](#plan-mode) for this message. The same as the Plan mode button. | +| `/compact` | Summarize the conversation so far and continue from that summary. See below. | +| `/feedback` | Report a bug or send feedback. Kai copies the debug details to your clipboard by default. Ask it to "open a ticket" and it opens the same support form as [Report a bug](#panel-header), pre-filled the same way. | + +### Compacting a long conversation + +Kai works from a limited amount of conversation at a time. When a chat gets long, `/compact` +replaces the earlier turns with a summary so there is room to keep going. Kai also compacts on its +own when a conversation grows too long, without you asking. + +**Your messages stay on screen, and that is intended.** Compaction adds a *Conversation compacted.* +line and removes nothing above it, so you keep the full transcript. + +What changes is what Kai reads. Everything above that line now reaches Kai as a summary, not as the +original messages. So you can scroll up and read a detail that Kai no longer has. + +Anything you type after the command steers the summary, which is worth doing when you know what +matters: + +``` +/compact keep the column mapping we worked out for the orders table +``` + +Compaction cannot be undone, so name what you need before you run it. For when to compact rather +than start a new chat, see [Manage Context](/kai/best-practices/#manage-context) in Best Practices. + +## Contextual Awareness + +Kai is aware of what you're currently viewing in the Keboola UI. Every message you send +includes your current page location, so Kai understands your context without needing +explicit references. + +- **Automatic context capture** — When you send a message, Kai receives your current URL path (e.g., which configuration, job, or table you're viewing) +- **Context-aware responses** — Kai uses this information to provide relevant suggestions and can reference "this configuration" or "the current job" naturally -This makes interactions more natural—you can say "analyze this job" while viewing a job, and Kai knows exactly which job you mean. When Kai investigates an issue across multiple configurations, you can follow along as it moves through your project. +This means you can say "analyze this job" while viewing a job, and Kai knows exactly +which job you mean. Turn on [Follow mode](#below-the-message-box) to watch it move through your project as +it works. ## Tips for New Users diff --git a/src/content/docs/kai/index.md b/src/content/docs/kai/index.md index 9de095965..de7f1a969 100644 --- a/src/content/docs/kai/index.md +++ b/src/content/docs/kai/index.md @@ -21,6 +21,8 @@ Kai is Keboola's embedded AI assistant—a context-aware data engineering co-pil **Data Modeling** — Build analytical frameworks, dimensional models, and complex data structures. +**Semantic Layer** - Build and maintain your project's [semantic layer](/ai/semantic-layer/) so every AI assistant shares the same definitions of your metrics and business terms. + ## Why Use Kai? **Context-aware** — Unlike generic AI tools, Kai reads your actual job logs, configurations, and data structures to provide specific solutions. @@ -53,6 +55,7 @@ Every user on a supported stack can see the **Kai Agent** button in the project' - [Use Cases & Examples](/kai/use-cases/) - [Best Practices](/kai/best-practices/) - [Python Client](/kai/python-client/) +- [Semantic Layer](/ai/semantic-layer/) ## Security & Privacy diff --git a/src/content/docs/kai/security-and-privacy.md b/src/content/docs/kai/security-and-privacy.md index ebef8fae9..2ac1922d1 100644 --- a/src/content/docs/kai/security-and-privacy.md +++ b/src/content/docs/kai/security-and-privacy.md @@ -9,22 +9,19 @@ Kai is designed with enterprise security in mind. This page explains how your da ## Your Data -**Never used for AI training** — Keboola does not use Kai inputs to train, retrain, or fine-tune models. +**Never used for AI training** — Keboola does not use Kai inputs to train, retrain, or fine-tune models. Keboola analyses de-identified usage data to improve Kai; it does not train models on your data. **Automatically deleted:** - Inference prompts and responses: Removed from Google within 30 seconds -- Observability logs (LangSmith, EU region): Deleted after 14 days. - We integrate LangSmith to continuously monitor, evaluate, and improve the assistant's quality, reliability, and user experience. - - PII and raw data content are redacted before traces are sent to LangSmith. Redaction covers names, emails, IDs, and other customer-specific values detected in either inputs or model responses. - - Tool responses containing data are never stored. - - Only scrubbed metadata and anonymized response summaries are stored for quality analysis. +- Code execution: E2B sandboxes are ephemeral and destroyed after each session; no customer data is retained +- Conversation logs: Kept by Keboola for up to 30 days, then de-identified - Your workspace data: Never leaves your Keboola project **Encrypted everywhere:** - In transit: TLS 1.2+ - At rest: AES-256 -**Regional processing** — Data is processed primarily in your chosen cloud region, with temporary routing to Google Vertex AI for real-time inference. +**Regional processing** — Prompts and responses are processed in your project's cloud region (EU or US) via Google Vertex AI. On EU stacks all LLM inference stays in the EU. Sandboxed code execution via E2B runs in the US; sandboxes are ephemeral and retain no customer data. ## AI Provider @@ -54,7 +51,7 @@ Contact [support@keboola.com](mailto:support@keboola.com) to discuss BYOLLM opti - **Never paste credentials in chat** — Kai uses secure configuration forms. Tell Kai what you need, and it will prompt you securely. - **Use development branches** — Test Kai-generated changes before merging to production. -- **Verify outputs** — AI-generated code may contain errors. Always review before deployment. +- **Verify outputs** — Kai is an AI and can make mistakes. Check anything important before deployment. ## Compliance @@ -68,8 +65,10 @@ Default AI provider (Google Vertex AI) maintains SOC 2 Type II, ISO 27001, GDPR, When using BYOLLM, compliance depends on your chosen provider and commercial agreement. +## Terms + +Your use of Kai is governed by Keboola's standard terms — the [Master Software Subscription Agreement](https://www.keboola.com/software-subscription-agreement) or the [Free Plan Terms of Services](https://www.keboola.com/free-plan-terms-and-conditions) — and by the [Data Processing Agreement](https://www.keboola.com/dpa). The sub-processors engaged for Kai (Google LLC, E2B Inc.) are listed at [security.keboola.com/subprocessors](https://security.keboola.com/subprocessors). + ## Questions? **Questions or concerns:** [support@keboola.com](mailto:support@keboola.com) - -For full details, see the [Kai AI Assistant Terms](https://www.keboola.com/ai-assistant-terms). diff --git a/src/content/docs/kai/settings.md b/src/content/docs/kai/settings.md index 24cfee97c..d76e34a6c 100644 --- a/src/content/docs/kai/settings.md +++ b/src/content/docs/kai/settings.md @@ -14,6 +14,8 @@ The settings panel has two tabs: **Tool Permissions** and **System Instructions* Tool Permissions let you control which tools Kai is allowed to use. This eliminates the need to manually approve each action — you can pre-approve tools you trust and block those you don't want Kai to use. +These are your own settings. They apply to your Kai in this project and change nothing for your teammates, who set their own. + ![Kai Settings — Tool Permissions](/kai/kai-settings-tool-permissions.png) ### Tool Categories @@ -27,11 +29,11 @@ Tools are organized into two categories: For each tool, you can set one of three permission levels: -| Permission | Behavior | -|------------|----------| -| **Always allow** | The tool runs automatically without asking for confirmation. | -| **Always ask** | Kai must request your approval each time before using the tool. | -| **Block** | The tool is completely disabled and Kai cannot use it. | +| | Permission | Behavior | +|--|------------|----------| +| ![Always allow](/kai/kai-perm-always-allow.png) | **Always allow** | The tool runs automatically without asking for confirmation. | +| ![Always ask](/kai/kai-perm-always-ask.png) | **Always ask** | Kai must request your approval each time before using the tool. | +| ![Block](/kai/kai-perm-block.png) | **Block** | The tool is completely disabled and Kai cannot use it. | ### Setting Permissions @@ -44,10 +46,16 @@ Your permissions persist across all conversations within the same project. ## System Instructions -System Instructions let you provide Kai with persistent context and guidelines so you don't have to repeat yourself in every chat. Instructions exist at two levels: **project-level** (shared across all users) and **user-level** (personal to you). User-level instructions amend project-level instructions — both are included in every conversation. +System instructions are standing rules Kai follows in every conversation, so you stop repeating yourself. "Always use snake_case." "Prefix staging tables with `stg_`." "Respond in German." + +They exist at two levels. **Project-level** instructions apply to everyone in the project. **User-level** instructions are personal to you and are added on top. Both are included in every conversation. + +Instructions are the place for short rules. For longer knowledge, such as data standards and project-wide conventions, use [context files](#context-files). For a step-by-step procedure you invoke when you need it, use [skill files](#skill-files). ### Project-Level Instructions +![Project-level system instructions in Settings → Kai Agent](/kai/kai-settings-project-instructions.png) + Project-level instructions apply to **all users** in the project. They are managed in the project settings: 1. Go to **Settings → Kai Agent** in the main Keboola navigation. @@ -58,12 +66,14 @@ Use project-level instructions for team-wide standards such as: - **Naming conventions** — e.g., "Always prefix staging tables with `stg_` and use snake_case for all column names." - **Coding standards** — e.g., "Write SQL transformations using CTEs instead of subqueries. Always include comments explaining business logic." -- **Project context** — e.g., "Our fiscal year starts in April. Revenue calculations should exclude returns and use the `completed_at` date." +- **Pipeline conventions** — e.g., "Load `in.c-*` buckets incrementally and never modify them in place. Build output tables from staging tables instead of one deep query." Project-level instructions can be edited by project admins and managers. ### User-Level Instructions +![User-level system instructions in the Kai chat panel](/kai/kai-settings-user-instructions.png) + User-level instructions are **personal to you** and are added on top of the project-level instructions. They are configured in the Kai chat panel: 1. Open the Kai chat panel. @@ -92,17 +102,19 @@ This means user-level instructions can refine or add to the project-level instru - Each instruction field supports up to **4,000 characters**. - Keep instructions clear and specific — vague guidelines are less effective. - Update instructions as your project evolves and conventions change. -- Focus on rules Kai can't infer from your project data alone (e.g., business logic, team preferences). +- Focus on rules Kai can't infer from your project data alone (e.g., team conventions and preferences). - If Kai doesn't seem to follow an instruction, try rephrasing it more directly. -- For knowledge that outgrows the 4,000-character limit — data standards documents, business glossaries — use [context files](#context-files) instead. +- For knowledge that outgrows the 4,000-character limit, such as data standards and project-wide conventions, use [context files](#context-files) instead. ## Context Files -Context files (also called knowledge files) are Markdown documents that Kai reads automatically at the start of every conversation. Use them to give Kai project knowledge that is too long for system instructions: data standards, naming conventions, business glossaries, or documentation of your data model. +Context files (also called knowledge files) are Markdown documents that Kai reads automatically at the start of every conversation. Use them to give Kai project knowledge that is too long for system instructions: data standards and project-wide conventions. Kai cannot open links for security reasons, so anything it needs to read has to arrive as a file. Reference material for a single task, such as the documentation for one external system, is better placed in a [skill file](#skill-files), which is loaded only when the skill runs. To manage them, go to **Settings → Kai Agent** in the main Keboola navigation and use the **Context files** card: -1. Click **Upload** and select a Markdown (`.md`) file. +![Context files card in Settings → Kai Agent](/kai/kai-settings-context-files.png) + +1. Drag a Markdown (`.md`) file onto the card, or click **Select Files**. 2. The file is uploaded and takes effect in every **new** conversation (running conversations are not affected). 3. To replace a file, upload the new version and delete the old one. @@ -114,6 +126,44 @@ Rules and limits: - A file named `CLAUDE.md` becomes Kai's top-level memory file; all other files are loaded as always-on rules alongside it. - Context files apply **project-wide** — every user's conversations include them. +### Example + +A context file is ordinary Markdown with no required structure. Give it headings and keep +it to things Kai cannot work out from the project itself: + +```markdown +# Data standards + +## Buckets + +- `in.c-*` holds raw extractor output. Never modify it directly. +- `out.c-*` holds tables other teams and BI tools read. + +## Naming + +- Staging tables take an `stg_` prefix. +- Columns are snake_case. +- Timestamps end in `_at` and are always UTC. + +## Transformations + +- Avoid deep chains of CTEs. Split a long query into intermediate staging tables and + assemble the final output table from those. +- Materialize anything more than one transformation reads instead of recomputing it. +- Filter and deduplicate before joining, not after. +- List columns explicitly when writing an output table. No `SELECT *` into a table + other teams depend on. +- An incremental load needs a primary key. Full loads are for + small lookup tables only. + +## Integrations + +- When a new integration is needed, prefer a Custom Python component over the + Generic Extractor. +``` + +If you upload only one file, name it `CLAUDE.md` so it becomes Kai's top-level memory file. + :::tip Every context file is read in every conversation, so keep the set small and focused. One well-structured standards document usually works better than many overlapping files. ::: @@ -122,9 +172,20 @@ Under the hood, context files are ordinary [Storage Files](/storage/files/) tagg ## Skill Files -Skills are reusable, on-demand playbooks that appear in the chat's **`/` slash-command menu** alongside Kai's built-in skills. Unlike context files, Kai loads a skill only when it is invoked — making skills the right place for longer, task-specific instructions (e.g., "build the monthly report," "onboard a new data source") that shouldn't consume context in every chat. +A skill is a playbook Kai runs when you need it. The skills you upload appear in the chat's **`/` slash-command menu**. Use them for longer, task-specific instructions such as "build the monthly report" or "onboard a new data source". Kai also carries built-in skills that it invokes on its own when they are relevant; those are not listed in the menu. -Manage them in **Settings → Kai Agent** using the **Skill files** card. Two formats are accepted: +Kai skills use the open [Agent Skills](https://agentskills.io/home) format: a Markdown file with `name` and `description` frontmatter, optionally packaged with the supporting files it references. + +:::tip[Difference between a Skill and a Context file] +A context file is read in every conversation and it can take up your context window space. +A skill is used only when called, or when your request matches the skill's description. +::: + +Manage them in **Settings → Kai Agent** using the **Skill files** card. + +![Skill files card in Settings → Kai Agent](/kai/kai-settings-skill-files.png) + +Two formats are accepted: 1. **A single `.md` file** starting with YAML frontmatter. The `name` and `description` fields are required — the description tells Kai when to invoke the skill: @@ -149,6 +210,49 @@ Rules and limits: Skill files are Storage Files tagged **`kai-skill`**. +### How a skill gets invoked + +You can call a skill in two ways: + +- **Call it yourself** by typing `/` in the chat and picking it from the menu. +- **Let Kai call it** when your request matches the skill's `description`. + +![Calling a skill from the slash-command menu](/kai/kai-skill-slash-menu.png) + +Kai reads only each skill's `name` and `description` up front, then loads the body when it +decides the skill applies (progressive disclosure). That makes the `description` decisive +for skills Kai invokes itself. Write what it does, then when to use it, in the words your +team actually types: + +```yaml +description: Build narrative, scroll-driven data stories where scroll position drives + charts, color, and animation. Use whenever the user wants a "scrollytelling" app, a + "data story", a "narrative dashboard", or wants to turn a dataset into a guided + scrolling experience instead of an explore-it-yourself dashboard. +``` + +Compare that with `description: Data stories`, which gives Kai nothing to match on. + +### What to write a skill for + +Write a skill when you want Kai to do something a particular way, every time. + +- **House style.** Your brand palette, layout conventions and component choices, so every + data app someone builds looks like it belongs to your company. +- **A recurring procedure.** Month-end close, onboarding a new data source, the same set of + quality checks. +- **A specialised output** Kai would not produce by default, where the instructions run to + pages rather than paragraphs. This is what `.skill` archives are for: a `SKILL.md` plus + reference files it can read when needed. + +Skills are project-wide, so one person can encode the standard once and the whole team gets +it. + +For writing the skill itself, the Agent Skills project publishes +[best practices for skill creators](https://agentskills.io/skill-creation/best-practices): +how to structure `SKILL.md`, how long to make it, and when to move detail into separate +reference files. + ## Managing Files via API or CLI Because context and skill files are ordinary Storage Files identified by a tag (`kai-context` or `kai-skill`), any Storage API client can manage them. Upload with the tag and the **permanent** flag (so the file never expires): diff --git a/src/content/docs/kai/use-cases.md b/src/content/docs/kai/use-cases.md index 0a47061f7..310b2bae8 100644 --- a/src/content/docs/kai/use-cases.md +++ b/src/content/docs/kai/use-cases.md @@ -81,6 +81,27 @@ Kai reads actual job logs, checks configurations, and traces data lineage to pro "Generate descriptions for all tables in the customer_data bucket." ``` +## Semantic Layer + +**Build a model:** +``` +"Build a semantic model from the curated tables in the out.c-sales bucket." +``` + +**Add a metric:** +``` +"Add a net revenue metric to the sales model: gross revenue minus refunds and discounts." +``` + +**Share it:** +``` +"Share the sales semantic model read-only with our finance project." +``` + +Kai explores your tables, drafts the definitions, and validates them before asking for your +approval to write anything. See [Semantic Layer](/ai/semantic-layer/) for what a model contains +and for the other ways to build one. + ## Data Exploration **Project overview:** diff --git a/src/content/docs/management/project/limits/index.md b/src/content/docs/management/project/limits/index.md index d66eb605c..294fc05d3 100644 --- a/src/content/docs/management/project/limits/index.md +++ b/src/content/docs/management/project/limits/index.md @@ -1,6 +1,7 @@ --- title: Project Limits slug: 'management/project/limits' +description: Business and platform limits of a Keboola project - time credits (PPU) per job type, storage size, and platform quotas. redirect_from: - /management/limits/ --- @@ -43,7 +44,7 @@ Measured in milliseconds, presented in hours (1 hour = 3,600 seconds). Every job based on - elapsed time of the job (in seconds), -- types of jobs (workspace, SQL, Python, and R transformations), and +- types of jobs (workspaces, SQL, Python, R, DuckDB, and dbt transformations), and - backend performance: XSmall, Small, Medium, Large. Types: @@ -51,6 +52,19 @@ Types: - **SQL** - PPU used for SQL transformation jobs, in case the customer is using [BYODB](/storage/byodb) and has a separate product for SQL jobs. - **CDC** - PPU used for CDC extractor jobs, in case the customer is using [CDC](/components/extractors/database/#change-data-capture-cdc) and has a separate product for CDC jobs. +**Which job type does your work bill as?** + +| What you run | Billed as | +|---------------------------------------------------------------------------|-------------------------------------| +| Snowflake or BigQuery transformation, SQL workspace, Query Service (JDBC) | SQL job / workspace / Query service | +| Python, R, or DuckDB transformation; Python or R (JupyterLab) workspace | Data Science job / workspace | +| dbt transformation | dbt job | +| Data app | DataApps | +| Data source component | Data source job | +| Data destination component | Data destination job | + +The other job types in the table below are named for the feature that produces them. + Below you will find an overview of time credits consumed by individual Keboola job types. If you need more information, please contact your CSM. @@ -93,10 +107,10 @@ If you need more information, please contact your CSM. | SMALL (SQL) | Snowflake SMALL DWH or equivalent | | MEDIUM (SQL) | Snowflake MEDIUM DWH | | LARGE (SQL) | Snowflake LARGE DWH | -| XSMALL (Python,R, Components) | 8 GB RAM, 1 CPU cores, 150GB SSD, shared | -| SMALL (Python,R, Components, DataApp) | 16 GB RAM, 2 CPU cores, 150GB SSD, shared | -| MEDIUM (Python,R, Components) | 32 GB RAM, 4 CPU cores, 150GB SSD, shared | -| LARGE (Python,R, Components) | 114 GB RAM, 14 CPU cores, 1TB SSD, dedicated | +| XSMALL (Python, R, DuckDB, Components) | 8 GB RAM, 1 CPU cores, 150GB SSD, shared | +| SMALL (Python, R, DuckDB, Components, DataApp) | 16 GB RAM, 2 CPU cores, 150GB SSD, shared | +| MEDIUM (Python, R, DuckDB, Components) | 32 GB RAM, 4 CPU cores, 150GB SSD, shared | +| LARGE (Python, R, DuckDB, Components) | 114 GB RAM, 14 CPU cores, 1TB SSD, dedicated | | SMALL (dbt) | Snowflake SMALL DWH or equivalent | | REMOTE (dbt) | Using user's remote DWH | diff --git a/src/content/docs/overview/onboarding/cheat-sheet/index.md b/src/content/docs/overview/onboarding/cheat-sheet/index.md index 888ba9562..c55980357 100644 --- a/src/content/docs/overview/onboarding/cheat-sheet/index.md +++ b/src/content/docs/overview/onboarding/cheat-sheet/index.md @@ -133,7 +133,7 @@ Listing columns enhances your query's safety and readability, ensuring you're cl :::tip[Safe Development Practices] Use development branches for any changes or new additions to ensure a controlled and safe development environment. This method keeps your work organized and -secure. For more on development branches, see this [guide](/getting-started/branches/). +secure. For more on development branches, see this [guide](/components/branches/tutorial/). ::: ## Automating Your Flow diff --git a/src/content/docs/storage/tables/csv-files.md b/src/content/docs/storage/tables/csv-files.md index be2a82757..edd4e5cc9 100644 --- a/src/content/docs/storage/tables/csv-files.md +++ b/src/content/docs/storage/tables/csv-files.md @@ -51,7 +51,7 @@ A CSV file in this format can be exported from - OpenOffice / LibreOffice Calc, where you simply save the file in a Text CSV file and select *Unicode (UTF-8)* encoding. - Google Drive, where it is the default output format (note, however, that you might - prefer to use the [Google Sheets data source connector](/getting-started/load/googlesheets/) instead). + prefer to use the [Google Sheets data source connector](/components/extractors/storage/google-drive/) instead). - Microsoft Excel by following the below instructions. ### Exporting from Microsoft Excel diff --git a/src/content/docs/transformations/duckdb/index.md b/src/content/docs/transformations/duckdb/index.md index d592a89bf..103c3c7e1 100644 --- a/src/content/docs/transformations/duckdb/index.md +++ b/src/content/docs/transformations/duckdb/index.md @@ -75,6 +75,13 @@ These actions are available from the transformation configuration page and are h ## Dynamic Backends +DuckDB runs in-process on the **data science backend** — the same container infrastructure that runs +Python and R transformations — not on a SQL data warehouse. Two things follow from this, and both +affect cost: + +- A DuckDB job is billed as a **Data Science job**, not as a SQL job, even though you write SQL in it. +- The backend size sets the memory of that container, not the size of a warehouse. + You can change the backend size to allocate more memory for your transformation. The following sizes are available: | Backend Size | Memory | Recommended For | @@ -86,6 +93,10 @@ You can change the backend size to allocate more memory for your transformation. Start with the **Small** backend and scale up as needed based on your dataset size and query complexity. +For the credit rate of each size, see the **Data Science job / workspace** rows in +[Project Limits](/management/project/limits/#project-power--time-credits). Do not use the **SQL job** +rates to estimate DuckDB cost — those apply to Snowflake and BigQuery transformations. + ***Note:** Dynamic backends are not available if you are on the [Free Plan (Pay As You Go)](/management/payg-project/).* ### Auto-Resource Detection @@ -348,7 +359,7 @@ FROM "pipeline_stages"; **Choose DuckDB for:** - Ad-hoc analysis and small to medium datasets - Rapid prototyping of transformations -- Projects with limited budgets +- Projects with limited budgets — DuckDB bills at data science rates, far below SQL warehouse rates (see [Dynamic Backends](#dynamic-backends)) - Datasets under a few terabytes - Development and testing diff --git a/src/content/docs/transformations/index.md b/src/content/docs/transformations/index.md index 5d82db759..7bc11e204 100644 --- a/src/content/docs/transformations/index.md +++ b/src/content/docs/transformations/index.md @@ -59,8 +59,9 @@ tables from the input mapping are taken, modified, and produced into the tables A backend is the engine running the transformation script. It is a database server ([Snowflake](https://www.snowflake.com/), -[BigQuery](https://cloud.google.com/bigquery), -[DuckDB](https://duckdb.org/)), +[BigQuery](https://cloud.google.com/bigquery)), +an in-process database engine +([DuckDB](https://duckdb.org/)), or a language interpreter ([Python](https://www.python.org/about/), [R](https://www.r-project.org/about.html)). diff --git a/src/content/docs/transformations/mappings/index.md b/src/content/docs/transformations/mappings/index.md index 249a5dc2d..520855cb1 100644 --- a/src/content/docs/transformations/mappings/index.md +++ b/src/content/docs/transformations/mappings/index.md @@ -254,7 +254,7 @@ malformed files, etc.) or, for example, to work with pre-trained models that you - **Processed Tags** *(deprecated, do not use)* — legacy option that assigned tags to input files after the transformation finished, used to process files incrementally. :::caution -*Processed Tags* and *Query* are deprecated and not compatible with [development branches](/getting-started/branches/). New configurations should not use them, and the UI no longer offers them. Existing configurations continue to work; affected projects will be contacted before any breaking change. +*Processed Tags* and *Query* are deprecated and not compatible with [development branches](/components/branches/tutorial/). New configurations should not use them, and the UI no longer offers them. Existing configurations continue to work; affected projects will be contacted before any breaking change. ::: #### Incremental file processing @@ -523,9 +523,9 @@ the Storage table metadata. Mismatched types may cause errors or silent data inc `SWAP TABLE` on Storage tables through Direct Mode output mapping. Schema changes must be done through [Storage](/storage/tables/). -**Do not use Direct Mode output mapping in development branches for production data.** Development branches -currently share the same Snowflake schema as production. Writing via Direct Mode output mapping in a dev branch -**will modify production data**. This limitation is being addressed in a future release. +**Direct Mode output mapping in development branches requires the [Branched storage](/components/branches/) feature.** Without the Branched storage Development branches +share the same Snowflake schema as production. Writing via Direct Mode output mapping in a development branch +**will modify production data**. #### Limitations @@ -538,7 +538,7 @@ currently share the same Snowflake schema as production. Writing via Direct Mode partially written state. Wrap related operations in transactions to mitigate this. - **Limited auditability** — Only an import event is created on success, compared to the more detailed event trail of standard output mapping. -- **Development branch isolation not supported** — Dev branch writes affect production data +- **Development branch isolation requires [Branched storage](/components/branches/) feature** — Without the feature Dev branch writes affect production data on Snowflake. Use caution when testing. - **Read-only, external, and linked buckets are not supported** — Direct Mode output mapping cannot write to buckets that are read-only, external schemas, or linked from another project. diff --git a/src/content/docs/transformations/python-plain/index.md b/src/content/docs/transformations/python-plain/index.md index c79124f88..3d3ec2d98 100644 --- a/src/content/docs/transformations/python-plain/index.md +++ b/src/content/docs/transformations/python-plain/index.md @@ -93,6 +93,8 @@ To speed it up, you can change the backend size in the configuration. Python tra ![Screenshot - Backend size configuration](/transformations/python-plain/backend-size.png) Scaling up the backend size allocates more resources to speed up your transformation, which impacts [time credits consumption](/management/project/limits/#project-power--time-credits). +Python transformations run on the data science backend, so they are billed at the +**Data Science job / workspace** rates, not the SQL job rates. ***Note:** Dynamic backends are not available to you if you are on the [Free Plan (Pay As You Go)](/management/payg-project/).* diff --git a/src/content/docs/transformations/r-plain/index.md b/src/content/docs/transformations/r-plain/index.md index b4fd5a2b1..56b418dca 100644 --- a/src/content/docs/transformations/r-plain/index.md +++ b/src/content/docs/transformations/r-plain/index.md @@ -99,6 +99,8 @@ To speed it up, you can change the backend size in the configuration. R transfor - Large Scaling up the backend size allocates more resources to speed up your transformation, which impacts [time credits consumption](/management/project/limits/#project-power--time-credits). +R transformations run on the data science backend, so they are billed at the +**Data Science job / workspace** rates, not the SQL job rates. ***Note:** Dynamic backends are not available to you if you are on the [Free Plan (Pay As You Go)](/management/payg-project/).* diff --git a/src/content/docs/getting-started/ad-hoc/index.md b/src/content/docs/workspace/ad-hoc-analysis/index.md similarity index 78% rename from src/content/docs/getting-started/ad-hoc/index.md rename to src/content/docs/workspace/ad-hoc-analysis/index.md index e2a0e7154..fcf6bbb74 100644 --- a/src/content/docs/getting-started/ad-hoc/index.md +++ b/src/content/docs/workspace/ad-hoc-analysis/index.md @@ -1,21 +1,22 @@ --- title: 'Ad-Hoc Data Analysis' -slug: 'getting-started/ad-hoc' +slug: 'workspace/ad-hoc-analysis' description: "Explore arbitrary data in a Python workspace: bring in a public BigQuery dataset, plot it in a Jupyter notebook, and install your own libraries." redirect_from: - /tutorial/ad-hoc/ + - /getting-started/ad-hoc/ --- +Not every question deserves a pipeline. When you want to poke at data once — plot it, model it, +throw it away — a **Python workspace** gives you a Jupyter notebook with your own libraries and no +obligation to productionize anything. This page works through one such exploration end to end: +pulling a public dataset in with the BigQuery connector, charting it in a notebook, and installing +a library the base image does not carry. -After you have loaded your tables, either [from a URL](/getting-started/load/) or -[using a data source connector](/getting-started/load/database/), [manipulated the data](/getting-started/transform/) in SQL, -written it [into Google Sheets](/getting-started/write/), and -set everything to run [automatically](/getting-started/automate/), let's take a look at some additional Keboola -features related to doing ad-hoc analysis. - -This part of the tutorial shows how to work with arbitrary data in Python -in a completely unrestricted way. Although our examples use the Python language, -the very same can be achieved using R. +Although the examples use Python, the very same can be achieved using R. If you have not created a +workspace before, start with [Create a Workspace](/workspace/create/); if you would rather build a +repeatable pipeline than explore, that is the +[Getting Started arc](/getting-started/) instead. Before you start, you should have a basic understanding of the [Python language](https://www.python.org/). @@ -51,7 +52,7 @@ Then create a [service account](https://cloud.google.com/iam/docs/service-accoun of the Google BigQuery data source connector, and create a Google Cloud Storage bucket as a temporary storage for off-loading the data from BigQuery. ***Note:** If setting up the Google BigQuery connector seems too complicated to you, export the query results to Google Sheets and -[load them from Google Sheets](/getting-started/load/googlesheets/). Or, export them to a CSV file, publish it, and [load it over HTTP](/getting-started/load/).* +[load them from Google Sheets](/components/extractors/storage/google-drive/). Or, export them to a CSV file, publish it, and [load it over HTTP](/getting-started/load/).* ### Prepare Before you start, have a Google service account and a Google Cloud Storage bucket ready. @@ -61,29 +62,29 @@ To create a Google service account, go to the [**Google Cloud Platform Console > IAM & admin > Service accounts**](https://console.cloud.google.com/iam-admin/serviceaccounts) and create a new service account: -![Screenshot - Google Service Account](/getting-started/ad-hoc/cloud-platform-service-account-1.png) +![Screenshot - Google Service Account](/workspace/ad-hoc-analysis/cloud-platform-service-account-1.png) Name the service account: -![Screenshot - Google Service Account Detail](/getting-started/ad-hoc/cloud-platform-service-account-3.png) +![Screenshot - Google Service Account Detail](/workspace/ad-hoc-analysis/cloud-platform-service-account-3.png) Grant the roles **BigQuery Data Editor**, **BigQuery Job User** and **Storage Object Admin** to your service account: -![Screenshot - Google Service Account Permissions](/getting-started/ad-hoc/cloud-platform-service-account-4.png) +![Screenshot - Google Service Account Permissions](/workspace/ad-hoc-analysis/cloud-platform-service-account-4.png) Finally, create a new JSON key and download it to your computer: -![Screenshot - Google Service Account Download](/getting-started/ad-hoc/cloud-platform-service-account-5.png) +![Screenshot - Google Service Account Download](/workspace/ad-hoc-analysis/cloud-platform-service-account-5.png) #### Google Cloud Storage bucket To create a Google Cloud Storage bucket, go to the [**Google Cloud Platform console > Storage**](https://console.cloud.google.com/storage/browser) and create a new bucket: -![Screenshot - Google Cloud Platform](/getting-started/ad-hoc/cloud-platform-storage-1.png) +![Screenshot - Google Cloud Platform](/workspace/ad-hoc-analysis/cloud-platform-storage-1.png) Enter the bucket's name and choose where to store your data (the location type *Region* is okay for our purpose): -![Screenshot - Create Bucket](/getting-started/ad-hoc/cloud-platform-storage-3.png) +![Screenshot - Create Bucket](/workspace/ad-hoc-analysis/cloud-platform-storage-3.png) Do not set a retention policy on the bucket. The bucket contains only temporary data and no retention is needed. @@ -91,39 +92,39 @@ Do not set a retention policy on the bucket. The bucket contains only temporary Now you're ready to load the data into Keboola. Go to the section **Components**, and click the green button **Add Component**: -![Screenshot - Components](/getting-started/ad-hoc/ex-bigquery-1.png) +![Screenshot - Components](/workspace/ad-hoc-analysis/ex-bigquery-1.png) Use the search to find the Google BigQuery data source: -![Screenshot - Add New Component search for Google BigQuery](/getting-started/ad-hoc/ex-bigquery-2.png) +![Screenshot - Add New Component search for Google BigQuery](/workspace/ad-hoc-analysis/ex-bigquery-2.png) Click **+ Add Component** and then **Connect To My Data**: -![Screenshot - New Configuration](/getting-started/ad-hoc/ex-bigquery-3.png) +![Screenshot - New Configuration](/workspace/ad-hoc-analysis/ex-bigquery-3.png) Name the configuration (e.g., 'Bls Unemployment') and describe it if you want. Then, click **Create Configuration**: -![Screenshot - New Configuration Name](/getting-started/ad-hoc/ex-bigquery-4.png) +![Screenshot - New Configuration Name](/workspace/ad-hoc-analysis/ex-bigquery-4.png) Then set the service account key: -![Screenshot - Big Query Authorization](/getting-started/ad-hoc/ex-bigquery-5.png) +![Screenshot - Big Query Authorization](/workspace/ad-hoc-analysis/ex-bigquery-5.png) Open the downloaded key you have created above in a text editor, copy & paste it in the input field, click **Submit** and then **Save**. -![Screenshot - Service Account Copy](/getting-started/ad-hoc/ex-bigquery-6.png) +![Screenshot - Service Account Copy](/workspace/ad-hoc-analysis/ex-bigquery-6.png) Fill the bucket you have created above: -![Screenshot - Big Query Unload](/getting-started/ad-hoc/ex-bigquery-7.png) +![Screenshot - Big Query Unload](/workspace/ad-hoc-analysis/ex-bigquery-7.png) After that configure the actual extraction queries by clicking the **Add Query** button: -![Screenshot - Big Query Configured](/getting-started/ad-hoc/ex-bigquery-8.png) +![Screenshot - Big Query Configured](/workspace/ad-hoc-analysis/ex-bigquery-8.png) Name the query, e.g., `Unemployment rates`: -![Screenshot - New Query Name](/getting-started/ad-hoc/ex-bigquery-9.png) +![Screenshot - New Query Name](/workspace/ad-hoc-analysis/ex-bigquery-9.png) Check *Create your own query using an SQL editor*, uncheck the *Use Legacy SQL* setting, and paste the following code in the *SQL Query* field: @@ -140,11 +141,11 @@ The `LNS14000000` series will pick the unemployment rates only. Then **Save** the query configuration. -![Screenshot - Query Configuration](/getting-started/ad-hoc/ex-bigquery-10.png) +![Screenshot - Query Configuration](/workspace/ad-hoc-analysis/ex-bigquery-10.png) Now run the configuration to bring the data to Keboola: -![Screenshot - Finished Configuration](/getting-started/ad-hoc/ex-bigquery-11.png) +![Screenshot - Finished Configuration](/workspace/ad-hoc-analysis/ex-bigquery-11.png) Running the data source connector creates a background job that @@ -167,25 +168,25 @@ To explore the data, go to [**Workspaces**](/workspace/). Provided for each user and project automatically, it is an isolated environment in which you can experiment without interfering with any production code. -![Screenshot - Transformations](/getting-started/ad-hoc/transformation-1.png) +![Screenshot - Transformations](/workspace/ad-hoc-analysis/transformation-1.png) Click on **New Sandbox** next to Python (Jupyter): -![Screenshot - Create Sandbox](/getting-started/ad-hoc/transformation-2.png) +![Screenshot - Create Sandbox](/workspace/ad-hoc-analysis/transformation-2.png) Select the unemployment rates table (`in.c-keboola-ex-google-bigquery-v2-548939034.unemployment-rates` in this case), click on **Create Sandbox**. Wait for the process to finish: -![Screenshot - Sandbox Configuration](/getting-started/ad-hoc/transformation-3.png) +![Screenshot - Sandbox Configuration](/workspace/ad-hoc-analysis/transformation-3.png) When finished, connect to the web version of the [Jupyter Notebook](https://jupyter.org/). It allows you to run arbitrary code by clicking the **Connect** button: -![Screenshot - Sandbox Credentials](/getting-started/ad-hoc/transformation-4.png) +![Screenshot - Sandbox Credentials](/workspace/ad-hoc-analysis/transformation-4.png) When prompted, enter the password from the Sandbox screen: -![Screenshot - Sandbox Login](/getting-started/ad-hoc/sandbox-1.png) +![Screenshot - Sandbox Login](/workspace/ad-hoc-analysis/sandbox-1.png) You can now run arbitrary code in Python, using common data scientist tools like [Pandas](https://pandas.pydata.org/) or [Matplotlib](https://matplotlib.org/). @@ -211,7 +212,7 @@ plt.suptitle('US Unemployment Rate', size=15) plt.show() ``` -![Screenshot - Sandbox Result](/getting-started/ad-hoc/sandbox-2.png) +![Screenshot - Sandbox Result](/workspace/ad-hoc-analysis/sandbox-2.png) ## Adding Libraries Now that you can experiment with the U.S. unemployment data extracted from Google BigQuery (or any other data extracted in any other way), @@ -258,7 +259,7 @@ plt.suptitle('EU Unemployment Rate', size=15) plt.show() ``` -![Screenshot - Sandbox Result](/getting-started/ad-hoc/sandbox-3.png) +![Screenshot - Sandbox Result](/workspace/ad-hoc-analysis/sandbox-3.png) ## Wrap Up You have just learnt to do a completely ad-hoc analysis of various data sets. If you need to run the above code regularly, diff --git a/src/content/docs/workspace/create/index.md b/src/content/docs/workspace/create/index.md new file mode 100644 index 000000000..b963e05b7 --- /dev/null +++ b/src/content/docs/workspace/create/index.md @@ -0,0 +1,65 @@ +--- +title: 'Create a Workspace' +slug: 'workspace/create' +description: 'Create a workspace — an isolated environment with a copy of your production data — to develop and test transformation code safely.' +redirect_from: + - /tutorial/manipulate/workspace/ + - /getting-started/transform/workspace/ +--- + +An integral aspect of creating a transformation is the development of the script itself. +In Keboola, you can use SQL, Python, or R by default. To simplify the process of writing these +scripts, we offer **workspaces** — a secure development and analytical environment where you can +interact with the data and develop your scripts with confidence. This page walks through creating +one; [Workspaces](/workspace/) covers what they are and how they behave in full. + +1. Navigate to **Workspaces** and click the **Create Workspace** button. + +![Create Workspace](/workspace/create/workspaces1.png) + +2. Select Snowflake SQL Workspace (or other SQL workspace depending on your project’s backend) + +![Select a Workspace](/workspace/create/workspaces2.png) + +3. Enter a *Name* and an optional *Description*. **Share with all project users** decides whether +colleagues can open the workspace too — leave it off for a personal playground. Click +**Create Workspace**. + +![Name the Workspace](/workspace/create/workspaces3.png) + +4. A creation job runs, and the workspace appears in the list with an **Active** badge. + +![Creating Job](/workspace/create/workspaces4.png) + +5. Click the workspace name to open its detail. + +![Access Details of the Job](/workspace/create/workspaces5.png) + + The banner at the top says **Read-Only Access to all project data is granted for this + workspace** — you can query every table in Storage straight away, with nothing to load first. + That is why **Load data** and **Unload data** in the right-hand menu are greyed out: they exist + for the other way of working, where you clone selected tables into the workspace and refresh + them yourself. **Workspace Parameters** records what you got — the backend, the backend size, + and the authentication type. + +![Workspace detail](/workspace/create/workspaces6.png) + +6. Click **Open SQL Editor** to query the data right in the browser. The editor lists every bucket + under **Storage Explorer**, and the result appears below the code block. + +![The SQL editor](/workspace/create/workspaces7.png) + + To use your own tools instead, follow **Set up SQL client connection** in the top banner — it + hands you a JDBC URL for DBeaver, DataGrip or VS Code through the open-source + [Keboola JDBC driver](/workspace/jdbc-driver/). + + +After completing the development of your queries, you can then copy and paste them into a +[transformation](/transformations/) configuration — which is exactly what +[Transform Data](/getting-started/transform/) does by hand in the Getting Started arc. + +## Going further + +- [Workspaces](/workspace/) — lifecycle, loading and unloading data, read-only input mapping. +- [SQL Editor](/workspace/sql-editor/) — query Storage without creating a workspace at all. +- [Keboola JDBC Driver](/workspace/jdbc-driver/) — connect your own IDE to a workspace. diff --git a/src/integrations/beacon-transforms.mjs b/src/integrations/beacon-transforms.mjs index 557259a50..1f4dabc06 100644 --- a/src/integrations/beacon-transforms.mjs +++ b/src/integrations/beacon-transforms.mjs @@ -557,6 +557,14 @@ function transformGlossaryList(tree) { if (!para?.children?.length) continue; const [term, ...rest] = para.children; if (!rest.length) continue; + // The em dash is only the delimiter that identifies this shape in Markdown. + // Once the list is a 2-col grid, the columns do the separating, so strip it — + // otherwise every definition renders as "— definition". + if (rest[0].type === 'text') { + rest[0].value = rest[0].value.replace(/^\s*(—|–|--|-)\s+/, ''); + if (!rest[0].value) rest.shift(); + } + if (!rest.length) continue; para.children = [ term, { diff --git a/src/sidebar.mjs b/src/sidebar.mjs index 2d5343935..59a612a59 100644 --- a/src/sidebar.mjs +++ b/src/sidebar.mjs @@ -41,29 +41,6 @@ export const sidebar = [ { slug: "getting-started/automate" }, { slug: "getting-started/app" }, { slug: "getting-started/next-steps" }, - { - label: "Going Further", - collapsed: true, - items: [ - { slug: "getting-started/load/googlesheets" }, - { slug: "getting-started/load/database" }, - { slug: "getting-started/transform/workspace" }, - { slug: "getting-started/ad-hoc" }, - { - label: "Development Branches", - collapsed: true, - items: [ - { label: "Overview", slug: "getting-started/branches" }, - { slug: "getting-started/branches/prepare-tables" }, - { slug: "getting-started/branches/prepare-files" }, - { slug: "getting-started/branches/tables-in-branch" }, - { slug: "getting-started/branches/files-in-branch" }, - { slug: "getting-started/branches/project-diff" }, - { slug: "getting-started/branches/merge-to-production" }, - ], - }, - ], - }, ], }, { @@ -481,6 +458,19 @@ export const sidebar = [ items: [ { label: "Overview", slug: "components/branches" }, { slug: "components/branches/merge-requests" }, + { + label: "Branches Tutorial", + collapsed: true, + items: [ + { label: "Overview", slug: "components/branches/tutorial" }, + { slug: "components/branches/tutorial/prepare-tables" }, + { slug: "components/branches/tutorial/prepare-files" }, + { slug: "components/branches/tutorial/tables-in-branch" }, + { slug: "components/branches/tutorial/files-in-branch" }, + { slug: "components/branches/tutorial/project-diff" }, + { slug: "components/branches/tutorial/merge-to-production" }, + ], + }, ], }, { slug: "components/ip-addresses" }, @@ -608,6 +598,8 @@ export const sidebar = [ items: [ { label: "Overview", slug: "workspace" }, { slug: "workspace/snowflake-workspaces-access-changes" }, + { slug: "workspace/create" }, + { slug: "workspace/ad-hoc-analysis" }, { slug: "workspace/sql-editor" }, { label: "Keboola JDBC Driver", @@ -665,6 +657,7 @@ export const sidebar = [ { label: "Overview", slug: "ai" }, { slug: "ai/ai-kit" }, { slug: "ai/mcp-server" }, + { slug: "ai/semantic-layer" }, ], }, { diff --git a/src/styles/custom.css b/src/styles/custom.css index 9935ab752..465b97b6b 100644 --- a/src/styles/custom.css +++ b/src/styles/custom.css @@ -1191,6 +1191,13 @@ nav.sidebar { height: auto; } +/* Standalone screenshots need more room than a run of text. Scoped to images that + are alone in their paragraph, so table icons (in ) and inline images are + untouched. */ +.sl-markdown-content p:has(> starlight-image-zoom-zoomable:only-child):not(:where(.not-content *)) { + margin-block: 2rem; +} + /* Mobile / tablet */ @media (max-width: 50rem) { .b-title-icon { width: 32px; height: 32px; font-size: 18px; }