diff --git a/_data/navigation.yml b/_data/navigation.yml index 2ec44eeb9..013376e4d 100644 --- a/_data/navigation.yml +++ b/_data/navigation.yml @@ -736,3 +736,35 @@ items: title: Run Orchestration - url: /automate/set-schedule/ title: Set Schedule + - url: /integrate/ + title: Integration + items: + - url: /integrate/storage/api/ + title: Storage API + items: + - url: /integrate/storage/api/configurations/ + title: Configurations + - url: /integrate/storage/api/import-export/ + title: Import & Export + - url: /integrate/storage/api/importer/ + title: API Importer + - url: /integrate/storage/api/tde-exporter/ + title: TDE Exporter + - url: /integrate/storage/python-client/ + title: Storage API Python Client + - url: /integrate/storage/r-client/ + title: Storage API R Client + - url: /integrate/storage/php-client/ + title: Storage API PHP Client + - url: /integrate/storage/docker-cli-client/ + title: Storage API Docker CLI Client + - url: /integrate/variables/ + title: Variables + items: + - url: /integrate/variables/tutorial/ + title: Variables Tutorial + - url: /integrate/artifacts/ + title: Artifacts + items: + - url: /integrate/artifacts/tutorial/ + title: Artifacts Tutorial diff --git a/public/integrate/artifacts/artifacts-tutorial-1.png b/public/integrate/artifacts/artifacts-tutorial-1.png new file mode 100644 index 000000000..b35de9b1a Binary files /dev/null and b/public/integrate/artifacts/artifacts-tutorial-1.png differ diff --git a/public/integrate/artifacts/artifacts-tutorial-2.png b/public/integrate/artifacts/artifacts-tutorial-2.png new file mode 100644 index 000000000..3b78098be Binary files /dev/null and b/public/integrate/artifacts/artifacts-tutorial-2.png differ diff --git a/public/integrate/artifacts/artifacts-tutorial-3.png b/public/integrate/artifacts/artifacts-tutorial-3.png new file mode 100644 index 000000000..d9ba47847 Binary files /dev/null and b/public/integrate/artifacts/artifacts-tutorial-3.png differ diff --git a/public/integrate/artifacts/artifacts-tutorial-4.png b/public/integrate/artifacts/artifacts-tutorial-4.png new file mode 100644 index 000000000..ae120cc96 Binary files /dev/null and b/public/integrate/artifacts/artifacts-tutorial-4.png differ diff --git a/public/integrate/storage/api/async-import-handling.svg b/public/integrate/storage/api/async-import-handling.svg new file mode 100644 index 000000000..777e93372 --- /dev/null +++ b/public/integrate/storage/api/async-import-handling.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/public/integrate/storage/new-table.csv b/public/integrate/storage/new-table.csv new file mode 100644 index 000000000..8dbb6c464 --- /dev/null +++ b/public/integrate/storage/new-table.csv @@ -0,0 +1,5 @@ +"id","secondCol" +"1","a" +"2","b" +"3","c" +"4","d" \ No newline at end of file diff --git a/public/integrate/variables/countries.csv b/public/integrate/variables/countries.csv new file mode 100644 index 000000000..3ff3dec77 --- /dev/null +++ b/public/integrate/variables/countries.csv @@ -0,0 +1,21 @@ +"COUNTRY","CARS" +"Belgium","6293781" +"Finland","3358232" +"Italy","41393877" +"Romania","6541260" +"Turkey","20193915" +"Bulgaria","2823705" +"France","38720798" +"Netherlands","8977994" +"Russia","42201083" +"Ukraine","8655700" +"Czech Republic","5116750" +"Germany","47418800" +"Poland","20671278" +"Spain","27528877" +"United Kingdom","33792233" +"Azerbaijan","1080912" +"Denmark","2723040" +"Hungary","3393075" +"Portugal","5650428" +"Sweden","5126572" diff --git a/public/integrate/variables/tutorial-1.png b/public/integrate/variables/tutorial-1.png new file mode 100644 index 000000000..736929496 Binary files /dev/null and b/public/integrate/variables/tutorial-1.png differ diff --git a/public/integrate/variables/tutorial-2.png b/public/integrate/variables/tutorial-2.png new file mode 100644 index 000000000..d0730e343 Binary files /dev/null and b/public/integrate/variables/tutorial-2.png differ diff --git a/public/integrate/variables/variables.svg b/public/integrate/variables/variables.svg new file mode 100644 index 000000000..5433c2a1c --- /dev/null +++ b/public/integrate/variables/variables.svg @@ -0,0 +1,3 @@ + + +
variables_id
variables_id
variable_values_id
variable_values_id
Main Configuration
(vendor.component)
Main Configuration...
Variables Configuration
(keboola.variables)
Variables Configurat...
config
config
variableValuesId
variableValuesId
Orchestration
Orchestration
Variable Values Row
Variable Values...
config
config
variableValuesId
variableValuesId
Run Orchestration
Run O...
config
config
variableValuesId
variableValuesId
Run Configuration
Run C...
Viewer does not support full SVG 1.1
\ No newline at end of file diff --git a/src/content/docs/ai/mcp-server/index.md b/src/content/docs/ai/mcp-server/index.md index 447dd286c..d28191196 100644 --- a/src/content/docs/ai/mcp-server/index.md +++ b/src/content/docs/ai/mcp-server/index.md @@ -3,6 +3,7 @@ title: Keboola Model Context Protocol (MCP) Server slug: 'ai/mcp-server' redirect_from: - /external-integrations/mcp-server/ + - /integrate/mcp/ --- :::caution diff --git a/src/content/docs/components/applications/triggers/dbt-cloud-job-trigger/index.md b/src/content/docs/components/applications/triggers/dbt-cloud-job-trigger/index.md index 7e6525da9..f104f5c96 100644 --- a/src/content/docs/components/applications/triggers/dbt-cloud-job-trigger/index.md +++ b/src/content/docs/components/applications/triggers/dbt-cloud-job-trigger/index.md @@ -26,6 +26,6 @@ If you check the **Wait for result** option, the component will wait for the job You can find out how to get a service account token in the [dbt Cloud documentation](https://docs.getdbt.com/docs/dbt-cloud-apis/service-tokens). ## Notes on Artifacts Usage -In order to be able to use Keboola artifacts, the project must have the ```artifact``` feature enabled. You can find more information about this in [Keboola's docs](https://developers.keboola.com/integrate/artifacts/). +In order to be able to use Keboola artifacts, the project must have the ```artifact``` feature enabled. You can find more information about this in [Keboola's docs](/integrate/artifacts/). diff --git a/src/content/docs/components/extractors/database/index.md b/src/content/docs/components/extractors/database/index.md index 7c873c636..dcac18fa8 100644 --- a/src/content/docs/components/extractors/database/index.md +++ b/src/content/docs/components/extractors/database/index.md @@ -3,6 +3,7 @@ title: Database Data Source Connectors slug: 'components/extractors/database' redirect_from: - /extractors/database/ + - /integrate/database/ --- diff --git a/src/content/docs/components/index.md b/src/content/docs/components/index.md index 4c2469ff8..99f52cba6 100644 --- a/src/content/docs/components/index.md +++ b/src/content/docs/components/index.md @@ -133,7 +133,7 @@ The bottom right panel shows a list of the configuration versions. Use the list - roll back to an older version. All of the operations can be [accessed via an API](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs). -The [developer guide](https://developers.keboola.com/integrate/storage/api/configurations/) explains how to work with configurations. +The [developer guide](/integrate/storage/api/configurations/) explains how to work with configurations. **Important**: Component configurations do not count towards your project quota. diff --git a/src/content/docs/data-apps/python-js/index.md b/src/content/docs/data-apps/python-js/index.md index 16a7aa048..a5932666c 100644 --- a/src/content/docs/data-apps/python-js/index.md +++ b/src/content/docs/data-apps/python-js/index.md @@ -327,7 +327,7 @@ def load_table(table_id: str) -> pd.DataFrame: return pd.read_csv(StringIO(response.text)) ``` -For a complete example using the official Python client library, see the [Keboola Storage Python Client documentation](https://developers.keboola.com/integrate/storage/python-client/). +For a complete example using the official Python client library, see the [Keboola Storage Python Client documentation](/integrate/storage/python-client/). ## Secrets and Environment Variables diff --git a/src/content/docs/flows/index.md b/src/content/docs/flows/index.md index c11a48045..c7e7efe8a 100644 --- a/src/content/docs/flows/index.md +++ b/src/content/docs/flows/index.md @@ -3,6 +3,7 @@ title: Conditional Flows slug: 'flows' redirect_from: - /flows/conditional-flows/ + - /integrate/orchestrator/ --- Flows allow you to build automated data pipelines with conditional logic, branching, retries, and robust error handling. You can define flows that react to the outcome of previous steps, dynamically control their next action, or even skip tasks entirely. diff --git a/src/content/docs/integrate/artifacts/artifacts-tutorial-1.png b/src/content/docs/integrate/artifacts/artifacts-tutorial-1.png new file mode 100644 index 000000000..b35de9b1a Binary files /dev/null and b/src/content/docs/integrate/artifacts/artifacts-tutorial-1.png differ diff --git a/src/content/docs/integrate/artifacts/artifacts-tutorial-2.png b/src/content/docs/integrate/artifacts/artifacts-tutorial-2.png new file mode 100644 index 000000000..3b78098be Binary files /dev/null and b/src/content/docs/integrate/artifacts/artifacts-tutorial-2.png differ diff --git a/src/content/docs/integrate/artifacts/artifacts-tutorial-3.png b/src/content/docs/integrate/artifacts/artifacts-tutorial-3.png new file mode 100644 index 000000000..d9ba47847 Binary files /dev/null and b/src/content/docs/integrate/artifacts/artifacts-tutorial-3.png differ diff --git a/src/content/docs/integrate/artifacts/artifacts-tutorial-4.png b/src/content/docs/integrate/artifacts/artifacts-tutorial-4.png new file mode 100644 index 000000000..ae120cc96 Binary files /dev/null and b/src/content/docs/integrate/artifacts/artifacts-tutorial-4.png differ diff --git a/src/content/docs/integrate/artifacts/index.md b/src/content/docs/integrate/artifacts/index.md new file mode 100644 index 000000000..bdf391ea1 --- /dev/null +++ b/src/content/docs/integrate/artifacts/index.md @@ -0,0 +1,129 @@ +--- +title: Artifacts +slug: 'integrate/artifacts' +--- + + +:::caution[Public Beta] +This is a preview feature and may change considerably in the future. The project must have the `artifacts` feature enabled. +::: + +**Artifacts** are additional files that can be produced or consumed by a [component](https://developers.keboola.com/extend/component). + +See the [Tutorial](/integrate/artifacts/tutorial/) for a step-by-step example. + +## Introduction +In some cases it's useful if a component not only extracts, transforms or uploads data, but also generate some other output, metadata or other runtime-discovered data. +These could be for example: +- AI models +- performance graphs of such models +- status updates from long-running tasks +- documentation +- data quality checks from in-progress tasks + +These additional information can be stored in artifacts and processed by another component or 3rd party tool. + +## Storage +Artifacts are stored in Keboola File Storage. + +## Types of artifacts +There are three types of artifacts for now `runs`, `custom` and `shared`. +The type specifies which components will have access to the artifact or which artifacts to download for the component to process. +Types are used in a configuration of a consumer component to specify which artifacts to download. + +- **runs** - artifacts from previous runs of the same configuration + +- **custom** - artifacts from previous runs of a different configuration. The configuration which produced the artifacts will be defined in the consumer configuration (configurationId, componentId, branchId) + +- **shared** - artifacts shared within an orchestration + +`runs` and `custom` types are the same from the producer point of view. To produce a `shared` artifact, it has to be written into a `shared` folder. Read more in [File structure](#file-structure) section. + +## File structure +Artifact is a unique set of files associated with a successful job, component and configuration. +A component can either produce or consume artifacts or both. + +### Produce +To produce an artifact, store one or more files in the following `output` directories. Subdirectories are also supported. +- `/data/artifacts/out/current` to create an artifact of type `runs` / `custom`. +- `/data/artifacts/out/shared` to create an artifact of type `shared`, which can be accessed by any component within the same orchestration. + +After the component job is finished all files and directories inside `current` and `shared` folders will be compressed into an archive and uploaded to File Storage with corresponding tags as a `artifact`. + +### Consume +To consume created artifacts you have to specify, in the configuration of a component, which artifacts (type) to download. + - `runs` to download artifacts produced by the same configuration and component. These will be stored in `/data/artifacts/in/runs/jobs/job-%job_id%` directory. + - `custom` to download artifacts produced by another configuration or component. These will be stored in `/data/artifacts/in/custom/jobs/job-%job_id%` directory. + - `shared` to download artifacts created within the same orchestration by any artifact producing component that has already finished. These will be stored in `/data/artifacts/in/shared/jobs/job-%job_id%` directory. + +## Configuration +Each type of artifact has a separate node in configuration. All the types can be used simultaneously. +Each type node has an attribute "enabled", which enables or disables download of the corresponding artifact type. + +### Runs + - **enabled** [true|false] - enable or disable download of this artifact type + - **filter** + - **date_since** - only artifacts from jobs younger than this will be downloaded + - **limit** - maximum number of the latest jobs from which to download artifacts + +### Custom +- **enabled** [true|false] - enable or disable download of this artifact type +- **filter** + - **branch_id**, **component_id**, **config_id** - specify the configuration to download artifacts from + - **date_since** - only artifacts from jobs younger than this will be downloaded + - **limit** - maximum number of the latest jobs from which to download artifacts + +### Shared +- **enabled** [true|false] - enable or disable download of this artifact type + +Full configuration example with all artifact types: + +```json +{ + "parameters": {}, + "artifacts": { + "runs": { + "enabled": true, + "filter": { + "date_since": "-7 days", + "limit": 5 + } + }, + "custom": { + "enabled": true, + "filter": { + "component_id": "keboola.python-transformation", + "config_id": "12345", + "branch_id": "default", + "date_since": "-7 days", + "limit": 5 + } + }, + "orchestration": { + "enabled": true + } + } +} +``` + +## Artifacts life-cycle in a job +Job runner checks if the project has enabled `artifacts` feature. +Job runner checks the configuration of the component. +If artifacts are enabled, it downloads artifacts to corresponding folders as configured (i.e. `runs`, `custom`, `shared`) and unzips them. + +Component process start and the component can: + +- access and process the downloaded artifacts in shared or custom directory + +- write artifacts to `current` or `shared` directory + +Component finishes and job runner does: + +- gzip the content of runs/current + +- tag the gzipped file with jobId, componentId, configId, runId, branchId and other tags if needed + +- upload the file to File Storage + +## File size limit +All the artifacts produced by a job shouldn’t be bigger than 1 GB. diff --git a/src/content/docs/integrate/artifacts/tutorial/index.md b/src/content/docs/integrate/artifacts/tutorial/index.md new file mode 100644 index 000000000..717b5316b --- /dev/null +++ b/src/content/docs/integrate/artifacts/tutorial/index.md @@ -0,0 +1,214 @@ +--- +title: Artifacts Tutorial +slug: 'integrate/artifacts/tutorial' +--- + + +This tutorial will show you how to work with artifacts. +In the following example we will use Python Transformation component to produce and consume artifacts. +But these principles would work inside any component. + +In the examples, we use the `curl` console tool to interact with our APIs. + +:::caution[Public Beta] +The `artifacts` feature must be enabled in your project. Contact [support@keboola.com](mailto:support@keboola.com) to enable it. + +The `artifacts` configuration can currently be created or edited only via the [Configuration API](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs). +::: + +## Examples + +For each example we will need [Storage API Token](/management/project/tokens/) to make the API call. + +1. Obtain a Storage API token from the user interface of your project, see this [Guide](/management/project/tokens/). +2. Store the token and url to the environment variable. + + ```shell + export STORAGE_API_HOST="https://connection.keboola.com" + export TOKEN="..." + ``` + +### 1. Produce artifact + +This is very simple example. We will just create a Python Transformation, which will write a file to the artifacts "upload" folder. +This file will be then uploaded as "artifact" to File Storage. + +1. In your Keboola project, create a new Python transformation, and paste this code into it: + ``` + import os + with open("/data/artifacts/out/current/myartifact1", "w") as file: + file.write("this is my artifact file content") + ``` + + ![Artifacts - transformation](/integrate/artifacts/artifacts-tutorial-1.png) + +2. Run the transformation - it should upload the file to File Storage as "artifact" + + ![Artifacts - Job](/integrate/artifacts/artifacts-tutorial-2.png) + +3. The file is now visible in File Storage with appropriate tags + + ![Artifacts - File Storage](/integrate/artifacts/artifacts-tutorial-2.png) + +### 2. Produce & consume artifacts + +To consume (download) artifacts for component to work with, we need to enable and configure artifacts download in the configuration of a component. + +We will create another configuration of the Python transformation via API. + +The artifacts part of the configuration will look like this. +It will enable download of artifacts of type `runs` with limit 5, which means this will download artifacts created by the last 5 runs of the same component configuration + + ```json + { + "artifacts":{ + "runs":{ + "enabled":true, + "filter":{ + "limit":5 + } + } + } + } + ``` + +The script of the transformation will look like following. +Files read from `/data/artifacts/in/runs/*/*` will be displayed at output - these are the artifact files downloaded. +The script will also generate a new artifact and write it to `/data/artifacts/out/current/myartifact1` as in previous example. + + ```python + import os + import glob + + # Download + print(glob.glob("/data/artifacts/in/runs/*/*")) + + # Upload + with open("/data/artifacts/out/current/myartifact1", "w") as file: + file.write("value1") + ``` +1. Run this curl command to create the configuration: + + ```shell + curl -X POST "$STORAGE_API_HOST/v2/storage/branch/default/components/keboola.python-transformation-v2/configs" \ + -H "X-StorageApi-Token: $TOKEN" \ + -H 'Content-Type: application/x-www-form-urlencoded' \ + --data-urlencode 'configuration={"parameters":{"blocks":[{"name":"Block 1","codes":[{"name":"artifacts","script":["import os\nimport glob\n\n# Download\nprint(glob.glob(\"/data/artifacts/in/runs/*/*\")) \n\n# Upload\nwith open(\"/data/artifacts/out/current/myartifact1\", \"w\") as file:\n file.write(\"value1\")"]}]}]},"artifacts":{"runs":{"enabled":true,"filter":{"limit":5}}}}' \ + --data-urlencode 'name=Artifacts upload & download' \ + --data-urlencode 'description=Test Artifacts upload & download' + ``` + +### 3. Consume artifacts from different component +Similar to previous example we will create a configuration of Python Transformation component. +But this time we will download artifacts produced by the configuration from `Example 2`. + +1. Export the id of the previously created configuration into an environment variable: + ```shell + export CONFIG_ID="..." + ``` + +2. Run curl command + ```shell + curl -X POST "$STORAGE_API_HOST/v2/storage/branch/default/components/keboola.python-transformation-v2/configs" \ + -H "X-StorageApi-Token: $TOKEN" \ + -H 'Content-Type: application/x-www-form-urlencoded' \ + --data-urlencode 'configuration={"parameters":{"blocks":[{"name":"Block 1","codes":[{"name":"artifacts","script":["import os\nimport glob\n\n# Download\nprint(glob.glob(\"/data/artifacts/in/custom/*/*\"))"]}]}]},"artifacts":{"custom":{"enabled":true,"component_id":"keboola.python-transformation","config_id":"$CONFIG_ID","branch_id":"default","filter":{"limit":5}}}}' \ + --data-urlencode 'name=Artifacts upload & download' \ + --data-urlencode 'description=Test Artifacts upload & download' + ``` + +The whole configuration now looks like this: + + ```json + { + "parameters": { + "blocks": [ + { + "name": "Block 1", + "codes": [ + { + "name": "artifacts", + "script": [ + "import os\nimport glob\n\n# Download\nprint(glob.glob(\"/data/artifacts/in/custom/*/*\"))" + ] + } + ] + } + ] + }, + "artifacts": { + "custom": { + "enabled": true, + "component_id": "keboola.python-transformation", + "config_id": "$CONFIG_ID", + "branch_id": "default", + "filter": { + "limit": 5 + } + } + } + } + ``` + +### 4. Shared artifacts +This example will show how to share artifacts within an orchestration +We will create two configurations of Python Transformation component. +One will produce a shared artifact and the other will consume it. +Both configurations needs to be in the same orchestration. +The configuration producing artifact needs to be in a phase that precedes the consuming one. + +1. Create "Producer" configuration + The Python code will write a file into a shared folder: + + ```python + import os + with open(path+\"/myartifact-shared\", \"w\") as file: + file.write(\"value1\")" + ``` + + Run curl command to create the configuration: + + ```shell + curl -X POST "$STORAGE_API_HOST/v2/storage/branch/default/components/keboola.python-transformation-v2/configs" \ + -H "X-StorageApi-Token: $TOKEN" \ + -H 'Content-Type: application/x-www-form-urlencoded' \ + --data-urlencode 'configuration={"parameters":{"blocks":[{"name":"Block 1","codes":[{"name":"Upload shared","script":["import os\npath = \"/data/artifacts/out/shared\"\nwith open(path+\"/myartifact3\", \"w\") as file:\n file.write(\"value1\")"]}]}]},"artifacts":{"runs":{"enabled":true,"filter":{"limit":5}}}}' \ + --data-urlencode 'name=Artifacts shared Producer' \ + --data-urlencode 'description=Artifacts upload shared' + ``` + +2. Create "Consumer" configuration + + The artifacts configuration: + ```json + { + "artifacts": { + "shared": { + "enabled": true + } + } + } + ``` + + The Python script: + + ```python + import os + import glob + print(glob.glob("/data/artifacts/in/shared/*/*")) + ``` + + Run curl command to create the configurtion: + + ```shell + curl -X POST "$STORAGE_API_HOST/v2/storage/branch/default/components/keboola.python-transformation-v2/configs" \ + -H "X-StorageApi-Token: $TOKEN" \ + -H 'Content-Type: application/x-www-form-urlencoded' \ + --data-urlencode 'configuration={"parameters":{"blocks":[{"name":"Block 1","codes":[{"name":"Download shared","script":["import os\nimport glob\n\nprint(glob.glob(\"/data/artifacts/in/shared/*/*\")) "]}]}]},"artifacts":{"shared":{"enabled":true}}}' \ + --data-urlencode 'name=Artifacts shared Consumer' \ + --data-urlencode 'description=Artifacts download shared' + ``` + +3. Now put each of the configurations into an Orchestration. "Artifacts shared Producer" into phase 1 and "Artifacts shared Consumer" into phase 2. + + ![Artifacts orchestration](/integrate/artifacts/artifacts-tutorial-4.png) diff --git a/src/content/docs/integrate/index.md b/src/content/docs/integrate/index.md new file mode 100644 index 000000000..09949f2cb --- /dev/null +++ b/src/content/docs/integrate/index.md @@ -0,0 +1,33 @@ +--- +title: Integration +slug: 'integrate' +--- + +You can look at Keboola as a system of independent and loosely coupled microservices (components). + +Each microservice has its own code base, and a publicly accessible API and configuration. +We do not cheat or have any advantage over other developers; our UI and other components use only these public APIs. + +As a result, it is very easy to, for example, write custom scripts to bootstrap a project, or do something that our UI does not offer. +Let's have a look into this! + +One of the very important components is [Storage](/storage/), which not only stores all data in a +project, but also provides additional functions such as managing other components and their configurations. +When you are integrating your systems with Keboola, **chances are that you want to start with [Storage](/storage/)**. + + \ No newline at end of file diff --git a/src/content/docs/integrate/storage/api/async-import-handling.svg b/src/content/docs/integrate/storage/api/async-import-handling.svg new file mode 100644 index 000000000..777e93372 --- /dev/null +++ b/src/content/docs/integrate/storage/api/async-import-handling.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/src/content/docs/integrate/storage/api/configurations/index.md b/src/content/docs/integrate/storage/api/configurations/index.md new file mode 100644 index 000000000..0b21dd20c --- /dev/null +++ b/src/content/docs/integrate/storage/api/configurations/index.md @@ -0,0 +1,523 @@ +--- +title: Component Configurations API +slug: 'integrate/storage/api/configurations' +--- + + +[Configurations](/components/) are an important part of a Keboola project. Most operations are +available in the UI. Use the API if you want to manipulate the configurations programmatically. + +Configurations represent component **instances** in a project. Each Keboola component has different configuration +options and requirements, which must be respected. As such, Keboola configurations provide a general framework for configuring +components, while the specific implementation details are left to the components themselves. + +When working with the [Component Configurations API](https://api.keboola.com/?service=storage#tag--Component-Configurations), +you need to know the `componentId` of the component being configured. +You can see a list of public components in [the Developer Portal](https://components.keboola.com/components), or you can get +a list of all available components with the [API index call](https://api.keboola.com/?service=storage#get-/v2/storage). +See our [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb). + +It will give you something like this: + +```json +{ + "host": "4edece0b0052", + "api": "storage", + "version": "v2", + "revision": "21fb56a0f6d61a307f350247a45950b1e4049625", + "documentation": "https://connection.keboola.com/api/storage/doc.json", + "components": [ + { + "id": "keboola.ex-aws-s3", + "type": "extractor", + "name": "AWS S3", + "description": "AWS Simple Storage Service", + "longDescription": "Download ... from AWS S3 and upload them to Storage.", + "version": 23, + "hasUI": false, + "hasRun": false, + "ico32": "https://ui.keboola-assets.com/.../keboola.ex-aws-s3/32/20.png", + "ico64": "https://ui.keboola-assets.com/.../keboola.ex-aws-s3/64/20.png", + "data": { + "definition": { + "type": "aws-ecr", + "uri": "147946154733.../keboola.ex-aws-s3", + "tag": "v3.0.0", + "repository": { + "region": "us-east-1" + } + }, + "vendor": { + "contact": [ + "Keboola", + "Křižíkova 488/115\n186 00 Prague 8\nCzech Republic", + "support@keboola.com" + ], + "licenseUrl": "https://github.com/keboola/aws-s3-extractor/blob/master/LICENSE" + }, + "configuration_format": "json", + "network": "bridge", + "memory": "512m", + "forward_token": false, + "forward_token_details": false, + "default_bucket": true, + "default_bucket_stage": "in", + "staging_storage": { + "input": "local" + } + }, + "flags": [ + "genericDockerUI", + "genericDockerUI-processors", + "appInfo.dataIn" + ], + "configurationSchema": {}, + "emptyConfiguration": {}, + "uiOptions": {}, + "configurationDescription": null, + "documentationUrl": "https://help.keboola.com/extractors/other/aws-s3/" + } + ], + "services": [...], + "urlTemplates": {...} +} +``` + +From here, you can see all available information about a particular component. In the following examples, we +will use `keboola.ex-aws-s3` --- the AWS S3 extractor. + +## Configuration Structure +Component configurations are largely dependent on the actual component being configured. This makes creating configurations manually +a bit tricky. Rather than starting from scratch, we recommend creating a configuration through the UI and then modifying it when you understand it. + +### Inspecting Configuration +To obtain an existing configuration, you can either use the list of configurations above or +the [Configuration Detail](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-) +API call. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) for obtaining a +configuration of the `keboola.ex-aws-s3` component. You will receive a response similar to this: + +```json +{ + "id": "364479526", + "name": "test", + "description": "", + "created": "2018-03-08T14:54:19+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "version": 5, + "changeDescription": "Table first table edited", + "isDeleted": false, + "configuration": { + "parameters": { + "accessKeyId": "AKIAIBZYEEXQILP46FCA", + "#secretAccessKey": "KBC::ComponentProjectEncrypted==p5gvUw4RSGiVJjT2ayVORpqS7yiKhExi7NnQECntVm8haHaHtFNVDMT8X8b+htnixpXhPIQ9yV+ETrvr+hNeYfh+Ex+UpC//QPWnLcEOC8XOLgmQN8BNgRGSERWUziK0" + } + }, + "rowsSortOrder": [], + "rows": [ + { + "id": "364481153", + "name": "first table", + "description": "", + "configuration": { + "parameters": { + "bucket": "travis-php-db-import-tests-s3filesbucket-vm9zhtm5jd7s", + "key": "tw_accounts.csv", + "saveAs": "first-table", + "includeSubfolders": false, + "newFilesOnly": true + }, + "processors": { + "after": [ + { + "definition": { + "component": "keboola.processor-move-files" + }, + "parameters": { + "direction": "tables", + "addCsvSuffix": true + } + }, + { + "definition": { + "component": "keboola.processor-create-manifest" + }, + "parameters": { + "delimiter": ",", + "enclosure": "\"", + "incremental": false, + "primary_key": [], + "columns": [], + "columns_from": "header" + } + }, + { + "definition": { + "component": "keboola.processor-skip-lines" + }, + "parameters": { + "lines": 1 + } + } + ] + } + }, + "isDisabled": false, + "version": 3, + "created": "2018-03-08T14:58:33+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited", + "state": { + "lastDownloadedFileTimestamp": "1511176959", + "processedFilesInLastTimestampSecond": [ + "tw_accounts.csv" + ] + } + } + ], + "state": {}, + "currentVersion": { + "created": "2018-03-08T23:27:37+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited" + } +} +``` + +The actual component configuration is split into three parts: + +- `configuration` node, containing an arbitrary component configuration +- `state` node, containing a component [state file](https://developers.keboola.com/extend/common-interface/config-file/#state-file) +- `rows` node, containing iterations of `configuration` and `state` + +The important part is the ID of the configuration you want to work with. In the following examples, we will use +`364479526`. + +### Configuration +The `configuration` node maps to the [configuration file](https://developers.keboola.com/extend/common-interface/config-file/#configuration-file-structure). +It can contain the `storage`, `parameters`, `processors` and `authorization` child nodes (the `image_parameters` and `action` nodes found in the config file +are injected at runtime and are not stored in the configuration). The `authorization` node is set in the configuration only when +[credentials injection](https://developers.keboola.com/extend/common-interface/oauth/#credentials-injection) should be used, otherwise it is also set during the runtime. +The `processors` node defines the [processors and their configuration](https://developers.keboola.com/extend/component/processors/). +The most common sub-nodes stored in the `configuration` node are therefore `parameters` (containing an arbitrary component configuration) +and `storage` (containing [input](https://developers.keboola.com/extend/component/tutorial/input-mapping/) and [output mapping](https://developers.keboola.com/extend/component/tutorial/output-mapping/)). +Both are transferred to the +configuration file without modification; that means that the [`storage` configuration](https://developers.keboola.com/extend/common-interface/config-file/#configuration-file-structure) +is directly usable in the `configuration` node. The `parameters` node is fully dependent on the component and has no universal specification or rules. + +In the above example, the `configuration` node contains the following: + +```json +"parameters": { + "accessKeyId": "AKIAIBZYEEXQILP46FCA", + "#secretAccessKey": "KBC::ComponentProjectEncrypted==p5gvUw4RSGiVJjT2ayVORpqS7yiKhExi7NnQECntVm8haHaHtFNVDMT8X8b+htnixpXhPIQ9yV+ETrvr+hNeYfh+Ex+UpC//QPWnLcEOC8XOLgmQN8BNgRGSERWUziK0" +} +``` + +That means that the component is not using input mapping nor output mapping. The allowed contents of `parameters` are described +in the [AWS S3 extractor code documentation](https://github.com/keboola/aws-s3-extractor#configuration-options). + +### Configuration Rows +The `rows` node contains iterations of the configuration. The interpretation of configuration rows is again dependent on the +component implementation. In the presented case of the `keboola.ex-aws-s3` component, each row corresponds to a single extracted table. +When `rows` node is non-empty, the component behavior is slightly modified. It behaves as if it were executed as many times as +there are rows. For each row, the `configuration` node from `root` and the `configuration` node from `rows` are merged, with +the latter overwriting the former in the case of conflict. + +Given the above configuration, the **effective configuration** passed to the component +[configuration file](https://developers.keboola.com/extend/common-interface/config-file/#configuration-file-structure) will be as follows: + +```json +{ + "parameters": { + "accessKeyId": "AKIAIBZYEEXQILP46FCA", + "#secretAccessKey": "KBC::ComponentProjectEncrypted==p5gvUw4RSGiVJjT2ayVORpqS7yiKhExi7NnQECntVm8haHaHtFNVDMT8X8b+htnixpXhPIQ9yV+ETrvr+hNeYfh+Ex+UpC//QPWnLcEOC8XOLgmQN8BNgRGSERWUziK0" + "bucket": "travis-php-db-import-tests-s3filesbucket-vm9zhtm5jd7s", + "key": "tw_accounts.csv", + "saveAs": "first-table", + "includeSubfolders": false, + "newFilesOnly": true + } +} +``` + +The first two parameters (`accessKeyId` and `#secretAccessKey`) are taken from the root `configuration`, the other +parameters are taken from the first rows' `configuration`. The `processors` node is never passed to the configuration file. +With the above configuration, the component will be executed only once, because there is one row. If there are no rows, the +component will still be executed once. If there were two rows, the component would be executed twice. + +If the component is executed more than once, the operations are executed in the following order: + +- input mapping for the first row +- run with the first row configuration (merged with root configuration) +- output mapping for the first row +- input mapping for the second row +- run with the second row configuration (merged with root configuration) +- output mapping for the second row + +All of these are executed in a single [job](https://developers.keboola.com/integrate/jobs/). However, even though multiple rows are executed in a single +job, the actual executions are still completely isolated. I.e., there is no way to share anything between the rows +(apart from the common `configuration`). It also means that the outputs of the first row are available in the Keboola project before +the second row starts, and the inputs for the second row are read only after the first row finishes processing. + +What is considered 'first' and 'second' -- i.e. the order of rows -- is defined by the order of items in the `rows` array. +See [below](#modifying-a-configuration) for an example of modifying the row order. + +Theoretically, configuration rows are supported for every component as long as the effective configuration matches what +the component expects. Configuration rows can be used to split the configuration into a common part (typically credentials) and an +iterable part which is repeated many times. Keep in mind that configurations heavily modified through the API might **not be supported +in the UI**. + +### State +The `state` node contains the content of the [state file](https://developers.keboola.com/extend/common-interface/config-file/#state-file). The +`state` is read from the state file and then supplied to the state file on the next run. In the above configuration, +the state is: + +```json +{ + "lastDownloadedFileTimestamp": "1511176959", + "processedFilesInLastTimestampSecond": [ + "tw_accounts.csv" + ] +} +``` + +`State` is considered an internal property of a component and you should avoid modifying it. The only reasonable modification of +`state` is to delete it -- in that case, the configuration will run as if it were run for the first time. To delete the `state`, set it to `{}`. +If configuration rows are used, then the `state` is stored separately for each row and the `state` node in configuration root is +not used. + +## Working with Configurations +Here, the most common operations done with configurations are described in examples. Feel free to go through the +[API reference](https://api.keboola.com/?service=storage#tag--Component-Configurations) for a full authoritative list of configuration features. + +### List Configurations +To obtain configuration details, use the [List Configs call](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs), +which will return all the configuration details. This means + +- the configuration itself (`configuration`) --- [section on configuration](#modifying-a-configuration) follows; +- configuration rows (`rows`) --- additional data of the configuration; and +- configuration state (`state`) --- [component state](https://developers.keboola.com/extend/common-interface/config-file/#state-file). + +Please note that the contents +of the `configuration`, `rows` and `state` sections depend purely on the component itself. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb). + +A sample result for the AWS S3 extractor looks like this: + +```json +[ + { + "id": "364479526", + "name": "test", + "description": "", + "created": "2018-03-08T14:54:19+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "version": 4, + "changeDescription": "Table first table edited", + "isDeleted": false, + "configuration": { + "parameters": { + "accessKeyId": "AKIAIBZYEEXQILP46FCA", + "#secretAccessKey": "KBC::ComponentProjectEncrypted==p5gvUw4RSGiVJjT2ayVORpqS7yiKhExi7NnQECntVm8haHaHtFNVDMT8X8b+htnixpXhPIQ9yV+ETrvr+hNeYfh+Ex+UpC//QPWnLcEOC8XOLgmQN8BNgRGSERWUziK0" + } + }, + "rowsSortOrder": [], + "rows": [ + { + "id": "364481153", + "name": "first table", + "description": "", + "configuration": {...}, + "isDisabled": false, + "version": 2, + "created": "2018-03-08T14:58:33+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited", + "state": {} + } + ], + "state": {}, + "currentVersion": { + "created": "2018-03-08T15:21:28+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited" + } + } +] +``` + +### Modifying Configuration +**Note: Configurations modified through the API might not be editable in the Keboola UI.** They can be run or used in an orchestration without any problems. + +Modifying a configuration means that a new version of that configuration is created. +For modifying a configuration, use the +[Update Configuration](https://api.keboola.com/?service=storage#put-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-) API call. +See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) in which the +configuration is modified to the following to set new credentials: + +```json +{ + "parameters": { + "accessKeyId": "a", + "#secretAccessKey": "b" + } +} +``` + +Notice that the configuration must be sent in the form field `configuration` as the endpoint does not accept pure JSON (yet). +Take great care to pass **only the contents** of the `configuration` node as in the above example. The configuration **must not be wrapped** in the +`configuration` node, otherwise the component will not +receive the configuration it expects. Also take care to properly escape the JSON using [URL encoding](https://en.wikipedia.org/wiki/Percent-encoding), +otherwise it may be misinterpreted. The raw HTTP request should look similar to this: + + curl --request PUT \ + --url https://connection.keboola.com/v2/storage/components/keboola.ex-aws-s3/configs/364479526 \ + --header "Content-Type: application/json" \ + --header 'X-StorageAPI-Token: {{token}}' \ + --data-binary "{ + \"configuration\": { + \"parameters\": { + \"accessKeyId\": \"a\", + \"#secretAccessKey\": \"b\" + } + } + }" + +Also note that the entire configuration must be always sent, there is no way to patch only part of it. +The same way the `configuration` is modified, other properties can be modified too. For example, you may want to +reset `state` by setting it to `{}`, or you can change the order of the configuration rows by setting the `rowsSortOrder` property. +The `rowsSortOrder` is an array of row ids -- see an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) (Set Row order of S3 extractor) +for the exact example request. + +### Modifying Configuration Row +Very similar to modifying a configuration, modifying a configuration **row** means that a new version of +the **entire configuration** is created. For modifying a configuration row, use the +[Update Row](https://api.keboola.com/?service=storage#put-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/rows/-rowId-) API call. + +See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) in which the +configuration row is modified to: + +```json +{ + "parameters": { + "bucket": "some-bucket", + "key": "sample.csv", + "includeSubfolders": false, + "newFilesOnly": true + } +} +``` + +The rules for updating a configuration row are the same as for [updating a configuration](#modifying-a-configuration). Also note that +a configuration row is never evaluated alone, it is always merged with the root `configuration`. If the same properties are defined +in the root `configuration` and row `configuration`, the values from the row are used. There is also an +[example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) of how to reset the row +state by setting `state` to `{}`. + +### Configuration Versions +When you [update a configuration](https://api.keboola.com/?service=storage#put-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-), +a new configuration version is actually created. In the above calls, only the last (active/published) configuration +is returned. To obtain a list of all recorded versions, use the +[List Versions API call](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/versions). +See this [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) +which would give you an output similar to the one below: + +```json +[ + { + "version": 4, + "created": "2018-03-08T15:21:28+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited", + "isDeleted": false, + "name": "test", + "description": "" + }, + { + "version": 3, + "created": "2018-03-08T14:58:33+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table added", + "isDeleted": false, + "name": "test", + "description": "" + }, + { + "version": 2, + "created": "2018-03-08T14:55:50+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "AWS Credentials edited", + "isDeleted": false, + "name": "test", + "description": "" + }, + { + "version": 1, + "created": "2018-03-08T14:54:19+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "", + "isDeleted": false, + "name": "test", + "description": "" + } +] +``` + +The field `version` represents the `version_id` in the following API example. + +### Rollback Configuration +After choosing a particular version, you can revert to that version by +[rolling back](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/versions/-versionId-/rollback), +i.e., making a new version identical to the chosen one. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D#2050856a-66b3-4120-9552-d1278a96621e) +of how to rollback the configuration `364479526` of the `keboola.ex-aws-s3` component to version `3`. + +It will create a new version of the configuration and return the ID of the version: +```json +{ + "version": "26" +} +``` + +### Creating Configuration Copy +After choosing a particular version, you can create a new independent +[configuration copy](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/versions/-versionId-/create) +of it. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) +of how to create a new configuration called `test-copy` from version `3` of the `364479526` configuration +for the `keboola.ex-aws-s3` component. + +It will return the ID of the newly created configuration: +```json +{ + "id": "364494012" +} +``` + diff --git a/src/content/docs/integrate/storage/api/import-export/index.md b/src/content/docs/integrate/storage/api/import-export/index.md new file mode 100644 index 000000000..57d5219fb --- /dev/null +++ b/src/content/docs/integrate/storage/api/import-export/index.md @@ -0,0 +1,336 @@ +--- +title: Manually Importing and Exporting Data +slug: 'integrate/storage/api/import-export' +--- + + +## Working with Data +Keboola Table Storage (Tables) and Keboola File Storage (File Uploads) are heavily connected together. +Keboola File Storage is technically a layer on top of the Amazon S3 service, and Keboola Table +Storage is a layer on top of a [database backend](/storage/#backends). + +To upload a table, take the following steps: + +- Request a [file upload](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/files/prepare) from +Keboola File Storage. You will be given a destination for the uploaded file on an S3 server. +- Upload the file there. When the upload is finished, the data file will be available in the *File Uploads* section. +- Initiate an [asynchronous table import](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async) +from the uploaded file (use it as the `dataFileId` parameter) into the destination table. +The import is asynchronous, so the request only creates a job and you need to poll for its results. +The imported files must conform to the [RFC4180 Specification](https://tools.ietf.org/html/rfc4180). + +![Schema of file upload process](/integrate/storage/api/async-import-handling.svg) + +Exporting a table from Storage is analogous to its importing. First, data is [asynchronously +exported](https://keboola.docs.apiary.io/#reference/tables/unload-data-asynchronously/asynchronous-export) from +Table Storage into File Uploads. Then you can request to [download +the file](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/files/-fileId-), which will give you +access to an S3 server for the actual file download. + +### Manually Uploading a File +To upload a file to Keboola File Storage, follow the instructions outlined in the +[API documentation](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/files/prepare). +First create a file resource; to create a new file called +[`new-file.csv`](/integrate/storage/new-table.csv) with `52` bytes, call: + +```bash +curl --request POST --header "Content-Type: application/json" --header "X-StorageApi-Token:storage-token" --data-binary "{ \"name\": \"new-file.csv\", \"sizeBytes\": 52, \"federationToken\": 1 }" https://connection.keboola.com/v2/storage/files/prepare +``` + +Which will return a response similar to this: + +```json +{ + "id": 192726698, + "created": "2016-06-22T10:44:35+0200", + "isPublic": false, + "isSliced": false, + "isEncrypted": false, + "name": "new_file2.csv", + "url": "https://s3.amazonaws.com/kbc-sapi-files/exp-15/1134/files/2016/06/22/192726697.new_file2?X-Amz-Content-Sha256=UNSIGNED-PAYLOAD&X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=AKIAJ2N244XSWYVVYVLQ%2F20160622%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date=20160622T084435Z&X-Amz-SignedHeaders=host&X-Amz-Expires=3600&X-Amz-Signature=86136cced74cdf919953cde9e2a0b837bd0b8f147aa6b7b30c2febde3b92d83d", + "region": "us-east-1", + "sizeBytes": 52, + "tags": [], + "maxAgeDays": 15, + "runId": null, + "runIds": [], + "creatorToken": { + "id": 53044, + "description": "ondrej.popelka@keboola.com" + }, + "uploadParams": { + "key": "exp-15/1134/files/2016/06/22/192726697.new_file2.csv", + "bucket": "kbc-sapi-files", + "acl": "private", + "credentials": { + "AccessKeyId": "ASI...H7Q", + "SecretAccessKey": "QbO...7qu", + "SessionToken": "Ago...bsF", + "Expiration": "2016-06-22T20:44:35+00:00" + } + } +} +``` + +The important parts are: `id` of the file, which will be needed later, the `uploadParams.credentials` node, +which gives you credentials to AWS S3 to upload your file, and +the `key` and `bucket` nodes, which define the target S3 destination as *s3://`bucket`/`key`*. +To upload the files to S3, you need an S3 client. There are a large number of clients available: +for example, use the +[S3 AWS command line client](https://docs.aws.amazon.com/cli/latest/userguide/cli-chap-install.html). +Before using it, [pass the credentials](https://docs.aws.amazon.com/cli/latest/topic/config-vars.html#credentials) +by executing, for instance, the following commands + +on *nix systems: +```bash +export AWS_ACCESS_KEY_ID=ASI...H7Q +export AWS_SECRET_ACCESS_KEY=QbO...7qu +export AWS_SESSION_TOKEN=Ago...wU= +``` + +or on Windows: +```bash +SET AWS_ACCESS_KEY_ID=ASI...H7Q +SET AWS_SECRET_ACCESS_KEY=QbO...7qu +SET AWS_SESSION_TOKEN=Ago...bsF +``` + +Then you can actually upload the `new-table.csv` file by executing the AWS S3 CLI [cp command](https://docs.aws.amazon.com/cli/latest/reference/s3/cp.html): +```bash +aws s3 cp new-table.csv s3://kbc-sapi-files/exp-15/1134/files/2016/06/22/192726697.new_file2.csv +``` + +After that, import the file into Table Storage, by calling either +[Create Table API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/buckets/-id-/tables-async) +(for a new table) or +[Load Data API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async) +(for an existing table). + +```bash +curl --request POST --header "Content-Type: application/json" --header "X-StorageApi-Token:storage-token" --data-binary "{ \"dataFileId\": 192726698, \"name\": \"new-table\" }" https://connection.keboola.com/v2/storage/buckets/in.c-main/tables-async +``` + +This will create an asynchronous job, importing data from the `192726698` file into the `new-table` destination table in the `in.c-main` bucket. +Then [poll for the job results](https://developers.keboola.com/integrate/jobs/#job-polling), or review its status in the UI. + +#### Python Example +The above process is implemented in the following example script in Python. This script uses the +[Requests](https://2.python-requests.org/en/master/) library for sending HTTP requests and +the [Boto 3](https://github.com/boto/boto3) library for working with Amazon S3. Both libraries can be +installed using pip: + +```bash +pip install boto3 +pip install requests +``` + +```python +import requests +import os +import json +import boto3 +from time import sleep + +storageToken = 'yourToken' +# Source filename (including path) +fileName = 'simple.csv' +# Target Storage Bucket (assumed to exist) +bucketName = 'in.c-main' +# Target Storage Table (assumed NOT to exist) +tableName = 'my-new-table' + +print('\nCreating upload file') + +# Create a new file in Storage +# See https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/files/prepare +response = requests.post( + 'https://connection.keboola.com/v2/storage/files/prepare', + data={ + 'name': fileName, + 'sizeBytes': os.stat(fileName).st_size, + 'federationToken': 1 + }, + headers={'X-StorageApi-Token': storageToken} +) +parsed = json.loads(response.content.decode('utf-8')) + +# Get AWS Credentials +accessKeyId = parsed['uploadParams']['credentials']['AccessKeyId'] +accessKeySecret = parsed['uploadParams']['credentials']['SecretAccessKey'] +sessionToken = parsed['uploadParams']['credentials']['SessionToken'] +region = parsed['region'] +fileId = parsed['id'] + +print('\nUploading to S3') + +# Upload file to S3 +# See https://boto3.amazonaws.com/v1/documentation/api/latest/guide/configuration.html +s3 = boto3.resource('s3', region_name=region, aws_access_key_id=accessKeyId, aws_secret_access_key=accessKeySecret, aws_session_token=sessionToken) +data = open(fileName, 'rb') +s3.Bucket(parsed['uploadParams']['bucket']).put_object(Key=parsed['uploadParams']['key'], Body=data) + +print('\nCreating table') + +# Load data from file into the Storage table +# See https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/buckets/-id-/tables-async +response = requests.post( + 'https://connection.keboola.com/v2/storage/buckets/%s/tables-async' % bucketName, + data={'name': tableName, 'dataFileId': fileId, 'delimiter': ',', 'enclosure': '"'}, + headers={'X-StorageApi-Token': storageToken}, +) +parsed = json.loads(response.content.decode('utf-8')) +if (parsed['status'] == 'error'): + print(parsed['error']) + exit(2) + +status = parsed['status'] +while (status == 'waiting') or (status == 'processing'): + print('\nWaiting for import to finish') + # See https://api.keboola.com/?service=storage#get-/v2/storage/jobs/-jobId- + response = requests.get(parsed['url'], headers={'X-StorageApi-Token': storageToken}) + jobParsed = json.loads(response.content.decode('utf-8')) + status = jobParsed['status'] + sleep(1) + +if (jobParsed['status'] == 'error'): + print(jobParsed['error']['message']) + exit(2) +``` + +#### Upload Files Using Storage API Importer +For production setup, we recommend using the approach [outlined above](#manually-uploading-a-file) +with direct upload to S3 as it is more reliable and universal. +In case you need to avoid using an S3 client, it is also possible to upload the +file by a simple HTTP request to [Storage API Importer Service](/integrate/storage/api/importer/). + +```bash +curl --request POST --header "X-StorageApi-Token:storage-token" --form "data=@new-file.csv" https://import.keboola.com/upload-file +``` + +The above will return a response similar to this: + +```json +{ + "id": 418137780, + "created": "2018-07-17T13:48:57+0200", + "isPublic": false, + "isSliced": false, + "isEncrypted": true, + "name": "404.md", + "url": "https:\/\/kbc-sapi-files.s3.amazonaws.com\/exp-15\/4088\/files\/2018\/07\/17\/418137779.new-file.csv...truncated", + "region": "us-east-1", + "sizeBytes": 1765, + "tags": [], + "maxAgeDays": 15, + "runId": null, + "runIds": [], + "creatorToken": { + "id": 144880, + "description": "file upload" + } +} +``` + +After that, import the file into Table Storage by calling either +[Create Table API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/buckets/-id-/tables-async) +(for a new table) or +[Load Data API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async) +(for an existing table). + +### Working with Sliced Files +Depending on the backend and table size, the data file may be sliced into chunks. +Requirements for uploading sliced files are described in the respective part of the +[API documentation](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/files/prepare). + +When you attempt to download a sliced file, you will instead obtain its manifest +listing the individual parts. Download the parts individually and join them +together. For a reference implementation of this process, see +our [TableExporter class](https://github.com/keboola/storage-api-php-client/blob/master/src/Keboola/StorageApi/TableExporter.php). + +**Important:** When exporting a table through the *Table* --- *Export* UI, the file will +be already merged and listed in the *File Uploads* section with the `storage-merged-export` tag. + +If you want to download a sliced file, [get credentials](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/files/-fileId-) +to download the file from AWS S3. Assuming that the file ID is 192611596, for example, call + +```bash +curl --header "X-StorageAPI-Token: storage-token" https://connection.keboola.com/v2/storage/files/192611596?federationToken=1 +``` + +which will return a response similar to this: + +```json +{ + "id": 192611596, + "created": "2016-06-21T15:25:35+0200", + "name": "in.c-redshift.blog-data.csv", + "url": "https://s3.amazonaws.com/kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csvmanifest?X-Amz-Content-Sha256=UNSIGNED-PAYLOAD&X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=AKIAJ2N244XSWYVVYVLQ%2F20160621%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date=20160621T135137Z&X-Amz-SignedHeaders=host&X-Amz-Expires=3600&X-Amz-Signature=ee69d94f0af06bcf924df0f710dcd92e6503a13c8a11a86be2606552bf9a8b26", + "region": "us-east-1", + "sizeBytes": 24541, + "tags": [ + "table-export" + ], + ... + "s3Path": { + "bucket": "kbc-sapi-files", + "key": "exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv" + }, + "credentials": { + "AccessKeyId": "ASI...UQQ", + "SecretAccessKey": "LHU...HAp", + "SessionToken": "Ago...uwU=", + "Expiration": "2016-06-22T01:51:37+00:00" + } +} +``` + +The field `url` contains the URL to the file manifest. Upon downloading it, you will get a JSON file with contents +similar to this: + +```json +{ + "entries": [ + {"url":"s3://kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv0000_part_00"}, + {"url":"s3://kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv0001_part_00"} + ] +} +``` + +Now you can download the actual data file slices. URLs are provided in the manifest file, and credentials to them +are returned as part of the previous file info call. To download the files from S3, you need an S3 client. There +are a wide number of clients available; for example, use the +[S3 AWS command line client](https://docs.aws.amazon.com/cli/latest/userguide/cli-chap-install.html). Before +using it, [pass the credentials](https://docs.aws.amazon.com/cli/latest/topic/config-vars.html#credentials) +by executing , for instance, the following commands + +on *nix systems: +```bash +export AWS_ACCESS_KEY_ID=ASI...UQQ +export AWS_SECRET_ACCESS_KEY=LHU...HAp +export AWS_SESSION_TOKEN=Ago...wU= +``` + +or on Windows: +```bash +SET AWS_ACCESS_KEY_ID=ASI...UQQ +SET AWS_SECRET_ACCESS_KEY=LHU...HAp +SET AWS_SESSION_TOKEN=Ago...wU= +``` + +Then you can actually download the files by executing the AWS S3 CLI [cp command](https://docs.aws.amazon.com/cli/latest/reference/s3/cp.html): +```bash +aws s3 cp s3://kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv0000_part_00 192611594.csv0000_part_00 +aws s3 cp s3://kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv0001_part_00 192611594.csv0001_part_00 +``` + +After that, merge the files together by executing the following commands + +on *nix systems: +```bash +cat 192611594.csv0000_part_00 192611594.csv0001_part_00 > merged.csv +``` + +or on Windows: +```bash +copy 192611594.csv0000_part_00 /B +192611594.csv0001_part_00 /B merged2.csv +``` diff --git a/src/content/docs/integrate/storage/api/importer/index.md b/src/content/docs/integrate/storage/api/importer/index.md new file mode 100644 index 000000000..95ce0a9b5 --- /dev/null +++ b/src/content/docs/integrate/storage/api/importer/index.md @@ -0,0 +1,48 @@ +--- +title: Storage API Importer +slug: 'integrate/storage/api/importer' +--- + + +The [whole process of importing](/integrate/storage/api/) a table into Storage can be simplified with the +Storage API Importer Service. +The Storage API Importer allows you to make an HTTP POST request and import a file directly into an existing Storage table. + +The HTTP request must contain the `tableId` and `data` form fields. The specified table must already exist in [Storage](/storage/). +Therefore to upload the `my-table.csv` CSV file (and replace the contents) into the `my-table` table in the `in.c-main` bucket, +call: + +```bash +curl --request POST --header "X-StorageApi-Token:storage-token" --form "tableId=in.c-main.my-table" --form "data=@my-table.csv" "https://import.keboola.com/write-table" +``` + +Using the Storage API Importer is the easiest way to upload data into Storage (except for +using one of the [API clients](/storage/)). However, the disadvantage is that the whole data file +has to be posted in a single HTTP request. **The maximum limit for a file size is 2GB and the transfer time is 45 minutes**. +This means that for substantially large files (usually more than hundreds of MB) +you may experience timeouts. If that happens, use the above outlined approach and upload the +files [directly to S3](/integrate/storage/api/import-export/#manually-uploading-a-file). + +## Parameters + +- `tableId` (required) Storage Table ID, example: in.c-main.users +- `data` (required) Uploaded CSV file. Raw file or compressed by [gzip](http://www.gzip.org/) +- `delimiter` (optional) Field delimiter used in a CSV file. The default value is ' , '. Use '\t' or type the tab char for tabulator. +- `enclosure` (optional) Field enclosure used in a CSV file. The default value is '"'. +- `escapedBy` (optional) CSV escape character; empty by default. +- `incremental` (optional) If incremental is set to 0 (its default), the target table is truncated before each import. + +Full list of avaialable parameters is available in the [API documentation](https://api.keboola.com/?service=import#import). + +## Examples +To load data incrementally (append new data to existing contents): + +```bash +curl --request POST --header "X-StorageApi-Token:storage-token" --form "incremental=1" --form "tableId=in.c-main.my-table" --form "data=@my-table.csv" "https://import.keboola.com/write-table" +``` + +To load data with a non-default delimiter (tabulator) and enclosure (empty): + +```bash +curl --request POST --header "X-StorageApi-Token:storage-token" --form "delimiter=\t" --form "enclosure=" --form "tableId=in.c-main.my-table" --form "data=@my-table.csv" "https://import.keboola.com/write-table" +``` diff --git a/src/content/docs/integrate/storage/api/index.md b/src/content/docs/integrate/storage/api/index.md new file mode 100644 index 000000000..71e9febd9 --- /dev/null +++ b/src/content/docs/integrate/storage/api/index.md @@ -0,0 +1,34 @@ +--- +title: Storage API +slug: 'integrate/storage/api' +--- + + +If you are new to Keboola, you should make yourself familiar with +the [Storage component](/storage/) before you start using it. +For a general introduction to working with Keboola APIs, see the [API Introduction](https://developers.keboola.com/overview/api/). +[Storage API](https://api.keboola.com/?service=storage) provides a number of functions. These are the most important ones: + +- [Component configurations](https://api.keboola.com/?service=storage#tag--Component-Configurations) +- [Storage tables](https://api.keboola.com/?service=storage#tag--Tables) +- [File uploads](https://api.keboola.com/?service=storage#tag--Files) +- [Storage buckets](https://api.keboola.com/?service=storage#tag--Buckets) + +Virtually, all API calls require a [Storage API token](/management/project/tokens/) to +be passed as the `X-StorageApi-Token` header. +Please note that the Storage API calls require the request to be sent +as `form-data` (unlike the rest of Keboola API, which is sent as `application/json`). + +For exporting tables from and importing tables to Storage, we highly recommend that you use one of the +[available clients](/storage/) or the [Storage API Importer service](/integrate/storage/api/importer/). +All imports and exports are done using CSV files. See +the [RFC4180 Specification](https://tools.ietf.org/html/rfc4180) for the format +and encoding specification, and +[User documentation](/storage/tables/csv-files/) for help on how to create such files. + +Continue reading the following sections for guidance on how to get started: + +- [Storage importer service for the easiest upload of data via API](/integrate/storage/api/importer/) +- [Getting started with component configurations](/integrate/storage/api/configurations/) +- [Importing and exporting data](/integrate/storage/api/import-export/) +- [TDE exporter for exporting data to Tableau Data Extracts](/integrate/storage/api/tde-exporter/) diff --git a/src/content/docs/integrate/storage/api/tde-exporter/index.md b/src/content/docs/integrate/storage/api/tde-exporter/index.md new file mode 100644 index 000000000..143edfdfb --- /dev/null +++ b/src/content/docs/integrate/storage/api/tde-exporter/index.md @@ -0,0 +1,93 @@ +--- +title: TDE Exporter +slug: 'integrate/storage/api/tde-exporter' +--- + + +[TDE Exporter](https://github.com/keboola/tde-exporter) exports tables from Keboola Storage into the +[TDE file format (Tableau Data Extract)](https://www.tableau.com/about/blog/2014/7/understanding-tableau-data-extracts-part1). +This component is normally a part of the [Tableau Writer](/tutorial/write/), +but it can also be used as a standalone component. + +Users can [run a TDE exporter job](https://developers.keboola.com/integrate/jobs/) as any other Keboola component or register it +as an orchestration task. After the exporter finishes, the resulting TDE files will be available in the +*Storage* --- *File uploads* section where you can download them via UI or [API](/integrate/storage/api/import-export/). + +## Running the Component +The TDE Exporter is a Keboola [component](https://developers.keboola.com/extend/component/) supporting both +[stored](/integrate/storage/api/configurations/) and +custom configurations supplied directly in the `run` request. + +### Stored Configuration +To run the TDE exporter with a stored configuration, first +[create the configuration](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs). +See [below](#custom-configuration) for the required configuration contents. +This call will give you the ID of the newly created configuration (for instance, `new-configuration-id`). +Then [create a job](https://developers.keboola.com/integrate/jobs/) with the specified configuration: + +```json +{ + "config": "new-configuration-id" +} +``` + +### Custom Configuration +You can specify the entire configuration in the API call. The JSON configuration conforms +to the [general configuration format](https://developers.keboola.com/extend/common-interface/config-file/). The specific part +is only the `parameters` section. A sample request to the `in.c-main.old-table` export table would look like this: + +```json +{ + "configData": { + "storage": { + "input": { + "tables": [{ + "source": "in.c-main.old-table" + }] + } + }, + "parameters": { + "tags": ["sometag"], + "typedefs": { + "in.c-main.old-table": { + "id": { + "type": "number" + }, + "col1": { + "type": "string" + } + } + } + } + } +} +``` + +The `parameters` section contains: + +- `tags`: array of tags that will be assigned to the resulting file in Storage File Uploads. +- `typedefs`: definitions of data types mapping source tables columns to destination TDE columns. + +The type definitions are entered as an object whose name must match the name of the table in the +`storage.input.tables.source` node (`in.c-main.old-table` in the above example). Object properties +are names of the table columns; each must have the `type` property which is one of the +[supported column types](https://help.tableau.com/current/pro/desktop/en-us/datafields_typesandroles_datatypes.htm): +`boolean`, `number`, `decimal`, `date`, `datetime` and `string`. + +## Date and DateTime +Data for these data types can be specified in the format used +in the [strptime function](https://pubs.opengroup.org/onlinepubs/009695399/functions/strptime.html). The format is specified as part of the column's type definition. For example: + +```json +{ + "col1": { + "type": "date", + "format":"%m-%d-%Y" + } +} +``` + +If no format is specified, the following default formats are used: + +- For `date`: `%Y-%m-%d` +- For `datetime`: `%Y-%m-%d %H:%M:%S or %Y-%m-%d %H:%M:%S.%f` diff --git a/src/content/docs/integrate/storage/docker-cli-client/index.md b/src/content/docs/integrate/storage/docker-cli-client/index.md new file mode 100644 index 000000000..417675924 --- /dev/null +++ b/src/content/docs/integrate/storage/docker-cli-client/index.md @@ -0,0 +1,111 @@ +--- +title: Storage Docker CLI Client +slug: 'integrate/storage/docker-cli-client' +redirect_from: + - /integrate/storage/php-cli-client/ +--- + + +The Storage API Docker command line interface (CLI) client is a portable command line client which provides +a simple implementation of [Storage API](https://api.keboola.com/?service=storage). +It runs on any platform which has Docker installed. + +Currently, the client implements + +- functions for exporting and importing tables; +- functions for creating and deleting buckets; and additionally, +- the [project backup feature](/management/project/export/). + +The client source is available in our [Github repository](https://github.com/keboola/storage-api-cli). +The client docker image is available in the [Quay repository](https://quay.io/repository/keboola/storage-api-cli?tab=tags). + +## Running in Docker +To print available commands: + +```bash +docker run quay.io/keboola/storage-api-cli:latest +``` + +The `latest` image tag always refers to the latest tagged version. + +## Running Phar + +PHAR (PHP Archive) is now deprecated, but there are still some older versions available. See the [repository documentation](https://github.com/keboola/storage-api-cli#running-phar-deprecated). + +### Example --- Creating a Table +To create a new table in Storage, use the `create-table` command. Provide the name of an +existing bucket, the name of the new table and a CSV file with the table's contents. + +To create the`new-table` table in the `in.c-main` bucket, use + +```bash +docker run --volume=$("pwd"):/data quay.io/keboola/storage-api-cli:latest create-table in.c-main new-table /data/new-table.csv --token=storage_token +``` + +or on Windows: + + docker run --volume=C:\Users\name\some-dir:/data quay.io/keboola/storage-api-cli:latest create-table in.c-main new-table /data/new-table.csv --token=storage_token + +or when using other then [default US region](https://developers.keboola.com/overview/api/#regions-and-endpoints), you need to provide the Storage API address: + +```bash +docker run --volume=$("pwd"):/data quay.io/keboola/storage-api-cli:latest create-table in.c-main new-table /data/new-table.csv --token=storage_token --url="https://connection.eu-central-1.keboola.com/" +``` + +Any of the above commands will import the contents of `new-table.csv` in the current directory into the newly +created table. You should see an output similar to this one: + + Authorized as: ondrej.popelka@keboola.com (Odinuv Sandbox) + Bucket found ok + Table create start + Table create end + Table id: in.c-main.new-table + +*Please note that the Docker container can only access folders within the container, so you need to mount a local folder. +In the example above, the local folder `$("pwd")` (replaced by the absolute path at runtime) is mounted as `/data` into the container. +The table is then accessible in this folder. The same approach applies to all other commands working with local files.* + +### Example --- Importing Data +If you only want to import new data into the table, use the `write-table` command and provide +the ID (*bucketName.tableName*) of an existing table. + +To import data into the `new-table` table in the `in.c-main` bucket, use + +```bash +docker run --volume=$("pwd"):/data quay.io/keboola/storage-api-cli:latest write-table in.c-main.new-table /data/new-data.csv --token=storage_token --incremental +``` + +The above command will import the contents of the `new-data.csv` file into the existing table. If the +`--incremental` parameter is supplied, the table contents will be appended. If the parameter is not +supplied, the table contents will be overwritten. You should see an output similar to this one: + + Authorized as: ondrej.popelka@keboola.com (Tutorial) + Table found ok + Import start + Import done in 17 secs. + + Results: + transaction: + warnings: + importedColumns: + - id + - secondCol + totalRowsCount: 8 + totalDataSizeBytes: 4096 + +### Example --- Exporting Data +If you want to export a table from Storage, use the `export-table` command. Provide +the ID (*bucketName.tableName*) of an existing table. + +To export data from the `old-table` table in the `in.c-main` bucket, use + +```bash +docker run --volume=$("pwd"):/data quay.io/keboola/storage-api-cli:latest export-table in.c-main.old-table /data/old-data.csv --token=storage_token +``` + +The above command will export the table from Storage and save it as `old-data.csv` in +the current directory. You should see an output similar to this one: + + Authorized as: ondrej.popelka@keboola.com (Tutorial) + Table found ok + Export done in 17 secs. diff --git a/src/content/docs/integrate/storage/php-client/index.md b/src/content/docs/integrate/storage/php-client/index.md new file mode 100644 index 000000000..f5c995d7b --- /dev/null +++ b/src/content/docs/integrate/storage/php-client/index.md @@ -0,0 +1,146 @@ +--- +title: Storage PHP Client Library +slug: 'integrate/storage/php-client' +--- + + +The Storage API PHP client library is a portable command line client providing +the most complete [Storage API](https://api.keboola.com/?service=storage) implementation. +It runs on any platform which has PHP installed. +Currently this client implements almost all Storage API functions including, of course, exporting and importing tables. + +The client source is available in our [Github repository](https://github.com/keboola/storage-api-php-client). + +## Installation + +The Library is available as a [Composer package](https://getcomposer.org/). +Unless you already have it, [install Composer](https://getcomposer.org/download/) on your system. +On *nix system, do so by running + +```bash +curl -s http://getcomposer.org/installer | php +mv ./composer.phar ~/bin/composer # or /usr/local/bin/composer +``` + +On Windows, use the [installer](https://getcomposer.org/Composer-Setup.exe). + +To install the library, run + +```bash +composer require keboola/storage-api-client +``` + +in the root of your project. You should get an output similar to this one: + + Using version ^4.11 for keboola/storage-api-client + ./composer.json has been created + Loading composer repositories with package information + Updating dependencies (including require-dev) + - Installing aws/aws-sdk-php (3.18.18) + Downloading: 100% + ... + - Installing keboola/storage-api-client (4.11.0) + Downloading: 100% + Writing lock file + Generating autoload files + +Then add the generated autoloader in your bootstrap script: + +```php +require 'vendor/autoload.php'; +``` + +You can read more in the [Composer documentation](https://getcomposer.org/doc/01-basic-usage.md). Packages +installable by Composer can be browsed at [Packagist package repository](https://packagist.org/). + +## Usage +The Storage API client is implemented as a single class. To create an instance of the class, provide a Storage API token to the +constructor. + +```php + 'your-token', +]); +``` + +### Example --- Create a Table +To create a new table in Storage, it is recommended to use an additional +[php-csv](https://github.com/keboola/php-csv) library to work +with CSV files. The library will get installed +automatically with the Storage API client, so you can use it out of the box. +To create a new table and import CSV data in it, use the following PHP script: + +```php + 'your-token', +]); +$csvFile = new CsvFile('./new-table.csv'); +$client->createTableAsync('in.c-main', 'new-table', $csvFile); +``` + +### Example --- Import Data +To import CSV data into an existing table and overwrite its contents, use the following PHP script: + +```php + 'your-token', +]); +$csvFile = new CsvFile('./new-table.csv'); +$client->writeTableAsync('in.c-main.new-table', $csvFile); +``` + +### Example --- Import Data Incrementally +To import CSV data into an existing table and append the new data to the existing table contents, use the following PHP script: + +```php + 'your-token', +]); +$csvFile = new CsvFile('./new-table.csv'); +$client->writeTableAsync('in.c-main.new-table', $csvFile, ['incremental' => true]); +``` + +All available upload options are listed in the [API documentation](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async). + +### Example --- Export Data +To export data from a Storage table to a CSV file, use the +`TableExporter` class. It is part of the client library. You can use the following script: + +```php + 'your-token' +]); + +$exporter = new TableExporter($client); +$exporter->exportTable('in.c-main.my-table', './old-table.csv'); +``` diff --git a/src/content/docs/integrate/storage/python-client/index.md b/src/content/docs/integrate/storage/python-client/index.md new file mode 100644 index 000000000..613a35a31 --- /dev/null +++ b/src/content/docs/integrate/storage/python-client/index.md @@ -0,0 +1,116 @@ +--- +title: Python Client Library +slug: 'integrate/storage/python-client' +--- + + +The Python client library is a [Storage API client](https://api.keboola.com/?service=storage) which you can use in your Python code. +The current implementation supports all basic data manipulations: + +- Importing data +- Exporting data +- Creating and deleting buckets and tables +- Creating and deleting workspaces + +The client source code is available in our [Github repository](https://github.com/keboola/sapi-python-client/). + +## Installation +This library is available on [Github](https://github.com/keboola/sapi-python-client), so we +recommend that you use the `pip` package to install it: + + pip3 install git+https://github.com/keboola/sapi-python-client.git + +## Usage +The client contains a `Client` class, which encapsulates all API endpoints and holds a storage token and URL. Each API endpoint is +represented by its own class (`Files`, `Buckets`, `Jobs`, etc.), which can be used standalone if you only work with one endpoint. +This means that the two following examples are equivalent: + +```python +from kbcstorage.client import Client + +client = Client('https://connection.keboola.com', 'your-token') +client.tables.detail('in.c-demo.some-table') +``` + +```python +from kbcstorage.tables import Tables + +tables = Tables('https://connection.keboola.com', 'your-token') +tables.detail('in.c-demo.some-table') +``` + +### Example --- Create Table and Import Data +To create a new table in Storage, use the `create` function of the `Tables` class. Provide the name of an existing bucket, +the name of the new table and a CSV file with the table's contents. + +To create the `new-table` table in the `in.c-main` bucket, use: + +```python +from kbcstorage.client import Client + +client = Client('https://connection.keboola.com', 'your-token') +client.tables.create(name='new-table', + bucket_id='in.c-main', + file_path='coords.csv', + primary_key=['id']) +``` + +The above command will import the contents of the `coords.csv` file into the newly created table. It will +also mark the `id` column as the primary key. +### Example --- Load to existing table, incrementally + +To load data incrementally into an existing table, we can use the [load](https://github.com/keboola/sapi-python-client/blob/5a93926c2191ccd6b7402c9e24d9912884d87d4c/kbcstorage/tables.py#L207) method, where `table_id` is the ID of the table that you want to load into, and `path` is the path to your csv file containing the data: + +```python + +from kbcstorage.client import Client + +client = Client('https://connection.keboola.com', 'your-token') + +client.tables.load(table_id=table_id, file_path=path, is_incremental=True) + +``` +### Example --- Export Data +To export data from the `old-table` table in the `in.c-main` bucket, use: + +```python +from kbcstorage.client import Client +import csv + +client = Client('https://connection.keboola.com', 'your-token') +client.tables.export_to_file(table_id='in.c-main.new-table', path_name='.') +with open('./new-table', mode='rt', encoding='utf-8') as in_file: + lazy_lines = (line.replace('\0', '') for line in in_file) + reader = csv.reader(lazy_lines, lineterminator='\n') + for row in reader: + print(row) +``` + +The above command will export the table from Storage into the file `new-table` and read it using +[CSV Reader](https://docs.python.org/3.6/library/csv.html#reader-objects). + +### Other Examples + +```python +# create a client +client = Client('https://connection.keboola.com', 'your-token') + +# create a bucket +client.buckets.create(name='demo', stage='in') + +# list buckets +client.buckets.list() + +# list all tables +client.tables.list() + +# list all tables in a bucket +client.buckets.list_tables(bucket_id='in.c-demo') + +# delete a table +client.tables.delete(table_id='in.c-demo.some-table') + +# delete a bucket +client.buckets.delete(bucket_id='in.c-main', force=True) + +``` diff --git a/src/content/docs/integrate/storage/r-client/index.md b/src/content/docs/integrate/storage/r-client/index.md new file mode 100644 index 000000000..e1c5b3c84 --- /dev/null +++ b/src/content/docs/integrate/storage/r-client/index.md @@ -0,0 +1,121 @@ +--- +title: R Client Library +slug: 'integrate/storage/r-client' +--- + + +The R client library is a [Storage API client](https://api.keboola.com/?service=storage) which you can use in your R code. +The current implementation supports all basic data manipulations: + +- Importing data +- Exporting data +- Creating and deleting buckets and tables + +The client source code is available in our [Github repository](https://github.com/keboola/sapi-r-client). + +## Installation +This library is available on [Github](https://github.com/keboola/sapi-r-client), so we +recommend that you use the `devtools` package to install it. + +```r +# first install the devtools package if it isn't already installed +install.packages("devtools") + +# install dependencies (another github package for aws requests) +devtools::install_github("cloudyr/aws.s3") + +# install the SAPI R client package +devtools::install_github("keboola/sapi-r-client") + +# load the library (dependencies will be loaded automatically) +library(keboola.sapi.r.client) +``` + +## Usage +To list available commands, run +```r +?keboola.sapi.r.client::SapiClient +``` + +**Important**: If you are running the code in R Studio, it might require a restart so that its help index is updated +and the above command works. + +The client is implemented as an [RC class](http://adv-r.had.co.nz/R5.html). To work with it, create an instance of the client. +The only required argument to create it is a valid Storage API token. + +```r +client <- SapiClient$new( + token = 'your-token' +) +``` + +### Example --- Create a Table and Import Data +To create a new table in Storage, use the `saveTable` function. Provide the name of an existing bucket, +the name of the new table and a CSV file with the table's contents. + +To create the `new-table` table in the `in.c-main` bucket, use + +```r +myDataFrame <- data.frame(id = c(1,2,3,4), secondCol = c('a', 'b', 'c', 'd')) +client <- SapiClient$new( + token = 'your-token' +) + +table <- client$saveTable( + df = myDataFrame, + bucket = "in.c-main", + tableName = "new-table", + options = list(primaryKey = 'id') +) +``` + +The above command will import the contents of the `myDataFrame` variable into the newly created table. It will +also mark the `id` column as the primary key. + +### Example --- Export Data +If you want to export a table from Storage and import it into R, use the `importTable` function. Provide +the ID (*bucketName.tableName*) of an existing table. + +To export data from the `old-table` table in the `in.c-main` bucket, use + +```r +client <- SapiClient$new( + token = 'your-token' +) + +data <- client$importTable('in.c-main.old-table') +``` + +The above command will export the table from Storage and save it in the `data` variable. The output is +a [data.table](https://cran.r-project.org/web/packages/data.table/index.html) object compatible with a `data.frame`. + +### Other Examples + +```r +# create a client +client <- SapiClient$new( + token = 'your-token' +) + +# verify the token +tokenDetails <- client$verifyToken() + +# create a bucket +bucket <- client$createBucket("new_bucket", "in", "A brand new Bucket!") + +# list buckets +buckets <- client$listBuckets() + +# list all tables +tables <- client$listTables() + +# list all tables in a bucket +tables <- client$listTables(bucket = bucket$id) + +# delete a table +client$deleteTable(table$id) + +# delete a bucket +client$deleteBucket(bucket$id) + +``` diff --git a/src/content/docs/integrate/variables/index.md b/src/content/docs/integrate/variables/index.md new file mode 100644 index 000000000..770137b17 --- /dev/null +++ b/src/content/docs/integrate/variables/index.md @@ -0,0 +1,731 @@ +--- +title: Variables +slug: 'integrate/variables' +--- + + +:::caution[Public Beta] +This is a preview feature and may change considerably in the future. +::: + +:::tip +Looking to set up variables through the Keboola UI? See [Variables in Transformations](/transformations/variables/). This page documents the underlying Configuration API used to define and resolve variables programmatically. +::: + +**Variables** are placeholders used in [configurations](/integrate/storage/api/configurations/). Their value is +resolved at [job runtime](https://developers.keboola.com/integrate/jobs/). + +**Important:** Make sure you're familiar with the [Configuration API](/integrate/storage/api/configurations/) and +the [Job API](https://developers.keboola.com/integrate/jobs/) before reading on. + +See the [Tutorial](/integrate/variables/tutorial/) for a step-by-step example. + +## Introduction +When using variables, the configuration is treated as a [Moustache template](https://mustache.github.io/mustache.5.html). +You can enter variables anywhere in the JSON of the configuration body. The configuration body is the contents of +the `configuration` node when you [retrieve a configuration](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-). +This means that you can't use variables in a name or in a configuration description. + +Variables are entered using the [Moustache syntax](https://mustache.github.io/mustache.5.html), +i.e., `{{ variableName }}`. To work with variables, three things are needed: + +- Main configuration -- the configuration in which variables are replaced (used); this can be a configuration of any component (e.g., a configuration of a transformation, extractor, writer, etc.). +- Variable configuration -- a configuration in which variables are defined; this is a configuration of a special `keboola.variables` component. +- Variable values -- actual values that will be placed in the main configuration. + +To enable replacement of variables, the *main configuration* has to reference the *variable configuration*. +If there is no *variable configuration* referenced, no replacement is made (the *main configuration* is completely +static). Variables can be used in any place of any configuration except legacy transformations (the component with +the ID `transformation`; it can still be used in a specific transformation -- e.g., `keboola.python-transformation-v2` +or `keboola.snowflake-transformation`, etc.), and an orchestrator (see [below](#orchestrator-integration)). + +## Variable Configuration +A *variable configuration* is a standard configuration tied to a special dedicated `keboola.variables` component. +The variable configuration defines names of variables to be replaced in the main configuration. You can create +the configuration using the +[Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs). +This is an example of the contents of such a configuration: + +```json +{ + "variables": [ + { + "name": "firstVariable", + "type": "string" + }, + { + "name": "secondVariable", + "type": "string" + } + ] +} +``` + +Note that `type` is always `string`. + +## Main Configuration +When you create a variable configuration, you'll obtain an ID of the configuration - e.g., `807940806`. +In the *main configuration*, you have to reference the *variable configuration* ID using the `variables_id` node. +Then you can use the variables in the configuration body: + +```json + +{ + "storage": { + "input": { + "tables": [ + { + "source": "in.c-application-testing.{{firstVariable}}", + "destination": "{{firstVariable}}.csv" + } + ] + }, + "output": { + "tables": [ + { + "source": "new-table.csv", + "destination": "out.c-transformation-test.cars" + } + ] + } + }, + "parameters": { + "script": [ + "print('{{firstVariable}}')" + ] + }, + "variables_id": "807940806" +} + +``` + +## Variable Values +You can either store the variable values as [configuration rows](/integrate/storage/api/configurations/#configuration-rows) of the +*variable configuration* and provide the row ID of the stored values at run time, or you can provide the variable values directly at run +time. There are three options how you can provide values to the variables: + +- Reference values using `variables_values_id` property in the *main configuration* (default values). +- Reference values using `variablesValuesId` property in job parameters. +- Provide values using `variableValuesData` property in job parameters. + +The structure of variable values, regardless of whether it is stored in configuration or provided at runtime, is as follows: + +```json +{ + "values": [ + { + "name": "firstVariable", + "value": "batman" + } + ] +} +``` + +## Variable Delimiter +The default variable delimiter is `{{` and `}}`. If the delimiter interferes with your code, it can +be changed as per the [Moustache docs](https://mustache.github.io/mustache.5.html). For example the +following piece of code + +```json +{ + "code": "SELECT \"COUNTRY\" || '{{ alias }}' || '{{=<< >>=}} {{ as-is }} <<={{ }}=>>' + AS \"COUNTRY\", \"CARS\" || '{{ size }}' AS \"CARS\" FROM \"my-table\"" +} +``` + +will be interpreted as (assuming the variables `alias=batman` and `size=big` are defined): + +```json +{ + "code": "SELECT \"COUNTRY\" || 'batman' || '{{ as-is }}' + AS \"COUNTRY\", \"CARS\" || 'big' AS \"CARS\" FROM \"my-table\"" +} +``` + +## Example Using Python Transformations +In this example, we will configure a Python transformation using variables. + +### Step 1 -- Create Variable Configuration +Use the [Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) +for the `keboola.variables` component with the following content: + +```json +{ + "variables": [ + { + "name": "alias", + "type": "string" + }, + { + "name": "size", + "type": "string" + } + ] +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#16a5d721-b6a4-4daa-9196-8e90250ed16b). + +### Step 2 -- Create Default Values for Variables +Note that this step is optional -- you can use variables without default values. +In the previous step, you obtained an ID of the variable configuration. Use the +[Create Configuration Row API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/rows). +Use the ID of the variable configuration and `keboola.variables` as a component. Use the following body: + +```json +{ + "values": [ + { + "name": "alias", + "value": "batman" + }, + { + "name": "size", + "value": "42" + } + ] +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#72de2851-1853-4fe3-bec3-856fcc9e2270) + +### Step 3 -- Create Main Configuration +Now it is time to create the actual configuration which will contain a Python transformation. +Use the following configuration body. The `storage` section describes the standard [input](https://developers.keboola.com/extend/common-interface/config-file/#input-mapping--basic) +and [output](https://developers.keboola.com/extend/common-interface/config-file/#output-mapping--basic) mapping. + +```json + +{ + "storage": { + "input": { + "tables": [ + { + "source": "in.c-variable-testing.{{alias}}", + "destination": "{{alias}}.csv" + } + ] + }, + "output": { + "tables": [ + { + "source": "new-table.csv", + "destination": "out.c-variable-testing.cars" + } + ] + } + }, + "parameters": { + "blocks": [ + { + "name": "First Block", + "codes": [ + { + "name": "First Code", + "script": [ + "import csv\ncsvlt = '\\n'\ncsvdel = ','\ncsvquo = '\"'\nwith open('in/tables/{{alias}}.csv', mode='rt', encoding='utf-8') as in_file, open('out/tables/new-table.csv', mode='wt', encoding='utf-8') as out_file:\n writer = csv.DictWriter(out_file, fieldnames=['COUNTRY', 'CARS'], lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo)\n writer.writeheader()\n\n lazy_lines = (line.replace('\\0', '') for line in in_file)\n reader = csv.DictReader(lazy_lines, lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo)\n for row in reader:\n writer.writerow({'COUNTRY': row['COUNTRY'] + '{{ alias }}', 'CARS': row['CARS'] + '{{ size }}'})\nfrom pathlib import Path\nimport sys\ncontents = Path('/data/config.json').read_text()\nprint(contents, file=sys.stdout)" + ] + } + ] + } + ] + } + "variables_id": "807968875", + "variables_values_id": "807952812" +} + +``` + +The `variables_id` property contains the ID of the [variable configuration](/integrate/variables/#step-1--create-variables-configuration) - e.g., `807968875`. The +`variables_values_id` property is optional and contains the ID of the [row with default values](/integrate/variables/#step-2--create-default-values-for-variable) - e.g., `807952812`. +The `parameters` section contains a script with the following Python code: + +```python + +import csv +csvlt = '\n' +csvdel = ',' +csvquo = '"' +with open('in/tables/{{alias}}.csv', mode='rt', encoding='utf-8') as in_file, open('out/tables/new-table.csv', mode='wt', encoding='utf-8') as out_file: + writer = csv.DictWriter(out_file, fieldnames=['COUNTRY', 'CARS'], lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo) + writer.writeheader() + lazy_lines = (line.replace('\0', '') for line in in_file) + reader = csv.DictReader(lazy_lines, lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo) + for row in reader: + writer.writerow({'COUNTRY': row['COUNTRY'] + '{{ alias }}', 'CARS': row['CARS'] + '{{ size }}'}) + +from pathlib import Path +import sys +contents = Path('/data/config.json').read_text() +print(contents, file=sys.stdout) + +``` + +The script reads a file given by the alias, modifies the two columns **COUNTRY** and **CARS**, and +prints the contents of the configuration file to output. + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#732e4b66-4f2d-46ab-80ba-7a7d07ddb94b). + +### Step 4 -- Run Job +There are three options for providing variable values when running a job: + +- Relying on default variables +- Providing ID of values using the `variablesValuesId` property in job parameters +- Providing values using the `variableValuesData` property in job parameters + +Following the rules for running a job, you always **have to** provide values for the defined variables. +Note that it is important which variables are *defined* in the variable configuration, not which +variables you actually use in the main configuration. For example, the main configuration references a variable +configuration with *firstVar* and *secondVar* variables, but you're using `{{ firstVar }}` and +`{{ thirdVar }}` in the configuration code. Then you have to provide values at least for *firstVar* +and *secondVar* variables. If you provide values for all *firstVar*, *secondVar*, and *thirdVar*, all of them will +be replaced. If you omit *thirdVar*, it will be replaced by an empty string. If you omit one of *firstVar*, +*secondVar*, an error will be raised. + +The second rule is that the three options of passing values are mutually exclusive. If you provide values using +`variablesValuesId` or `variableValuesData`, it overrides the default values (if provided). You can't use +`variablesValuesId` and `variableValuesData` together in a single call. If you do that, an error will be raised. +If no default values are set and none of the `variablesValuesId` or `variableValuesData` is provided, an error +will be raised. + +#### Option 1 -- Rely on default variables +If you created the default values, you can now directly run the job. Use the [Create Job API call](https://api.keboola.com/?service=job-queue#job-queue/tag/jobs/POST/jobs) +with the following body: + +```json +{ + "component": "keboola.python-transformation-v2", + "config": "807943784", + "mode": "run" +} +``` + +The `config` property contains the ID of the [main configuration](/integrate/variables/#step-3--create-main-configuration). +Before executing the API call, you have to create the source table. Unless you modified the mapping in the +[example](/integrate/variables/#step-3--create-main-configuration), you have to create a bucket named +**variable-testing** in the **in** stage. Then create a table called **batman** with columns **COUNTRY** +and **CARS**. You can use this [sample CSV file](/integrate/variables/countries.csv). + +After you create the input table, you can run the job. +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#31486ac2-ea52-4f19-a039-2ee1b1ae5863). +It will create a new table in Storage -- **out.c-variable-testing.cars**. The tables should contain the default +values, e.g.: + +|COUNTRY|CARS| +|---|---| +|Belgiumbatman|629378142| +|Finlandbatman|335823242| +|Italybatman|4139387742| +|Romaniabatman|654126042| + +The events of the job will contain the contents of the [configuration file](https://developers.keboola.com/extend/common-interface/config-file/) +where you can verify that the variables were replaced. + +
+ Click to expand the configuration. +```json +{ + "storage": { + "input": { + "tables": [ + { + "source": "in.c-variable-testing.batman", + "destination": "batman.csv", + "columns": [], + "where_values": [], + "where_operator": "eq" + } + ], + "files": [] + }, + "output": { + "tables": [ + { + "source": "new-table.csv", + "destination": "out.c-variable-testing.cars", + "incremental": false, + "primary_key": [], + "columns": [], + "delete_where_values": [], + "delete_where_operator": "eq", + "delimiter": ",", + "enclosure": "\"", + "metadata": [], + "column_metadata": [] + } + ], + "files": [] + } + }, + "parameters": { + "blocks": [ + { + "name": "First Block", + "codes": [ + { + "name": "First Code", + "script": [ + "import csv\ncsvlt = '\\n'\ncsvdel = ','\ncsvquo = '\"'\nwith open('in/tables/{{alias}}.csv', mode='rt', encoding='utf-8') as in_file, open('out/tables/new-table.csv', mode='wt', encoding='utf-8') as out_file:\n writer = csv.DictWriter(out_file, fieldnames=['COUNTRY', 'CARS'], lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo)\n writer.writeheader()\n\n lazy_lines = (line.replace('\\0', '') for line in in_file)\n reader = csv.DictReader(lazy_lines, lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo)\n for row in reader:\n writer.writerow({'COUNTRY': row['COUNTRY'] + '{{ alias }}', 'CARS': row['CARS'] + '{{ size }}'})\nfrom pathlib import Path\nimport sys\ncontents = Path('/data/config.json').read_text()\nprint(contents, file=sys.stdout)" + ] + } + ] + } + ] + }, + "variables_id": "807943784", + "variables_values_id": "807952812", + "image_parameters": {}, + "action": "run", + "authorization": {} +} +``` +
+ +#### Option 2 -- Run a job with stored values +Similarly to the [default values](http://localhost:4000/integrate/variables/#step-2--create-default-values-for-variable), +you can store another set of values. Let's add another configuration row to the *existing* variable configuration: + +```json +{ + "values": [ + { + "name": "alias", + "value": "WATMAN" + }, + { + "name": "size", + "value": "4200" + } + ] +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#fbe487b5-cd68-4318-8219-7c067ebef795). +You will obtain an ID of the row. Then create a table called **watman** with +columns **COUNTRY** and **CARS**. You can use this [sample CSV file](/integrate/variables/countries.csv). + +Run a job with parameters and provide the ID of the main configuration in the `config` property and +the ID of the value row in `variableValuesId`: + +```json +{ + "component": "keboola.python-transformation-v2", + "config": "807968875, + "mode": "run", + "variableValuesId": "807957572" +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#f883eb13-3f20-4e03-bf1b-36e9c889f773). +The output table now contains: + +|COUNTRY|CARS| +|---|---| +|BelgiumWATMAN|62937814200| +|FinlandWATMAN|33582324200| +|ItalyWATMAN|413938774200| + +#### Option 3 -- Run a job with inline values +The last option to provide the values for variables is to enter them directly when running a job. +Variable values are entered in the `variableValuesData` property: + +```json +{ + "component": "keboola.python-transformation-v2", + "config": "807968875", + "mode": "run", + "variableValuesData": { + "values": [ + { + "name": "alias", + "value": "batman" + }, + { + "name": "size", + "value": "scatman" + } + ] + } +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#2c38d6ca-2eda-4c7e-9888-071fad3d31d8). + +The output table will contain: + +|COUNTRY|CARS| +|---|---| +|Belgiumbatman|6293781scatman| +|Finlandbatman|3358232scatman| +|Italybatman|41393877scatman| + +## Orchestrator Integration +Variables in a configuration interact with an orchestrator in two ways: + +- Variables can be entered in task configuration. +- Variables can be entered when running an orchestration. + +Entering variable values in task configurations allows the orchestration to run configurations with variables. +Variable values are entered in the `actionParameters` property. The parameters are identical to +[running a job](/integrate/variables/#step-4--run-job). + +When running an orchestration, you can also provide variable values for an entire orchestration. In that case, +the provided values will override those set in individual orchestration tasks. The parameters are identical +to [running a job](/integrate/variables/#step-4--run-job). + +### Step 5 -- Create Orchestration +You have to use the +[Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) +to create a configuration of the `keboola.orchestrator` component. +You can use the following data in the configuration: + +```json +{ + "phases": [ + { + "id": 2468, + "name": "Extractors", + "dependsOn": [] + } + ], + "tasks": [ + { + "id": 13579, + "name": "Example", + "phase": 2468, + "task": { + "componentId": "keboola.python-transformation-v2", + "configId": "807968875", + "mode": "run", + "variableValuesId": "807952812" + }, + "continueOnFailure": false, + "enabled": true + } + ] +} +``` + +The contents of the `task` property are identical to the body +of the [run job API call](/integrate/variables/#step-4--run-job). Here, the value `807968875` refers to the ID +of the main configuration, and `807952812` refers to the ID of the configuration row with variable values. +You can use the `variableValuesData` field in the same manner. +Creating the above configuration will return a response containing the configuration ID, e.g., `807969959`. +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#9f2f9da0-59eb-4f33-a206-e5add24725d1). + +### Step 6 -- Run Orchestration +When running an orchestration which contains configurations referencing variables, you have to provide their +values. You can either rely on the stored values (either at the component configuration or in the orchestration task) +or you can provide the values at runtime. + +#### Option 1 -- Rely on stored values +Use the [Run Job API call](https://api.keboola.com/?service=job-queue#job-queue/tag/jobs/POST/jobs) +to run an orchestration. In its simplest form, the request body needs to contain just the ID of the orchestration +(obtained in the previous step): + +```json +{ + "component": "keboola.orchestrator", + "config": "807969959", + "mode": "run" +} +``` + +As long as the variable values can be found somewhere, this is sufficient. See [an example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#3ebdc3f5-a940-4f0d-860b-ec311f704a7e). + +#### Option 2 -- Provide values +Use the [Run Job API call](https://api.keboola.com/?service=job-queue#job-queue/tag/jobs/POST/jobs) +to run an orchestration. Additionally, you can use the `variableValuesId` or `variableValuesData` property +to override variable values set to individual tasks. The calling convention is the same as shown in the +[basic job run](/integrate/variables/#step-4--run-job). The same rules also apply, notably that you can't +use `variableValuesId` and `variableValuesData` together. +A sample request body: + +```json +{ + "component": "keboola.orchestrator", + "config": "807969959", + "mode": "run", + "variableValuesData": { + "values": [ + { + "name": "alias", + "value": "batman" + }, + { + "name": "size", + "value": "scatman" + } + ] + } +} +``` + +See [an example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#f4fcf7af-afbe-4c29-999e-0f4c50aa477b). + +## Variables Evaluation Sequence +There is a number of places where variable values can be provided (either as a reference to an existing row with +values or as an array of `values`): + +- Parameters in the orchestration +- Parameters in `task` setting of the orchestration +- Parameters in the component job itself +- Default values stored in configuration (`variables_values_id` property) + +The following diagram shows the parameters mentioned on this page and to what they refer to: + +![Screenshot -- Properties references](/integrate/variables/variables.svg) + +In a nutshell, `variableValuesId` always refers to the row of the variable configuration associated with the +main configuration. The main configuration is referenced in the `config` parameter. From another point of view, +the `config` parameter represents the configuration (either a component or an orchestration) to be run. +Note that in stored configurations snake_case is used instead of camelCase. + +The following rules describe the evaluation sequence: + +- Values provided in job parameters (a component job or an orchestration job) override the stored values. +- Values provided in an orchestration job override the stored values in `task`. +- Values provided in `task` override values stored in the component configuration. +- `variableValuesData` and `variableValuesId` can't be used together, so neither of them takes precedence. A reference to stored values can't be mixed with providing the values inline. +- If no values are provided anywhere, the default values are used. If no default values are present, an error is raised. + +## Shared Code +Related to variables is the Shared Code feature. Shared code allows to share parts of configuration code. In a +configuration it is also replaced using the [Moustache syntax](https://mustache.github.io/mustache.5.html). Shared code +is referenced using `shared_code_id` and `shared_code_row_ids` configuration nodes. Unlike variables, shared code can't +be overridden at runtime (so there are no parameters to set when running a job or in orchestration). +Shared code can, however, contain its own variables which need to be merged to those of the main configuration. + +### Creating Shared Code +Shared code pieces is stored as configuration rows of a dedicated component `keboola.shared-code`. Before creating a +piece of a shared code, you first have to create a configuration. Notice that the UI uses certain configurations for +certain components so you might want to check the existing configurations of `keboola.shared-code` component before +crating a new configuration. + +To create a configuration, use the [create configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs). The configuration content is ignored, i.e all you need to provide is name: + +```bash +curl --location --request POST 'https://connection.keboola.com/v2/storage/components/keboola.shared-code/configs' \ +--header 'X-StorageAPI-Token: my-token' \ +--header 'Content-Type: application/x-www-form-urlencoded' \ +--data-urlencode 'name=python-code' +``` + +Let's assume that the created configuration ID is `618884794`. +Next step is to create the shared code piece itself. To do this create a configuration row of the above configuration +with the configuration row content containing a piece of share code, for example: + +```json +{ + "code_content": [ + "from os import listdir\nfrom os.path import isfile, join\n\nmypath = '\''/data/in/files'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)\nmypath = '\''/data/in/user'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)" + ] +} +``` + +It is advisable to set a reasonable `rowId` of the row, because it will be used later to reference the shared code: + +```bash +curl --location --request POST 'https://connection.keboola.com/v2/storage/components/keboola.shared-code/configs/618884794/rows' \ +--header 'X-StorageApi-Token: my-token' \ +--header 'Content-Type: application/x-www-form-urlencoded' \ +--data-urlencode 'configuration={ + "code_content": ["from os import listdir\nfrom os.path import isfile, join\n\nmypath = '\''/data/in/files'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)\nmypath = '\''/data/in/user'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)"] +} +' \ +--data-urlencode 'rowId=dumpfiles' +``` + +The above example creates a piece of shared python code named `dumpfiles` which contains the +following python code: + +```python +from os import listdir +from os.path import isfile, join + +mypath = '/data/in/files' +onlyfiles = [f for f in listdir(mypath)] +print(onlyfiles) +mypath = '/data/in/user' +onlyfiles = [f for f in listdir(mypath)] +print(onlyfiles) +``` + +### Using Shared Code +To use a piece of shared code, you have to reference it in a configuration using `shared_code_id` which is the ID of the shared code configuration and `shared_code_row_ids` which is an array of IDS of shared code pieces. With the above example you need to add the following nodes to the configuration: + +```json +{ + "storage": {...}, + "parameters": {...}, + "shared_code_id": "618884794", + "shared_code_row_ids": ["dumpfiles"] +} +``` + +With that all moustache references to `{{ dumpfiles }}` will be replaced by the shared code piece. All other +moustache references will be kept untouched and be treated like variables. E.g: the following configuration: + +```json +{ + "storage": {}, + "parameters": { + "blocks": [ + { + "name": "Main block", + "codes": [ + { + "name": "Main code", + "script": ["{{ someOtherPlaceholder }}"] + }, + { + "name": "Debug", + "script": ["{{ dumpfiles }}"] + } + ] + } + ] + }, + "variables_id": "618878103", + "variables_values_id": "618878104", + "shared_code_id": "618884794", + "shared_code_row_ids": ["dumpfiles"] +} +``` + +Will be modified to: + +```json +{ + "storage": {}, + "parameters": { + "blocks": [ + { + "name": "Main block", + "codes": [ + { + "name": "Main code", + "script": ["{{ someOtherPlaceholder }}"] + }, + { + "name": "Debug", + "script": ["from os import listdir\nfrom os.path import isfile, join\n\nmypath = '\''/data/in/files'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)\nmypath = '\''/data/in/user'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)"] + } + ] + } + ] + }, + "variables_id": "618878103", + "variables_values_id": "618878104", + "shared_code_id": "618884794", + "shared_code_row_ids": ["dumpfiles"] +} +``` + +The variables then need to contain `someOtherPlaceholder` variable in order to produce a fully functional configuration. +The same way if the shared code piece contains any variables, they have to be set when running the configuration. + +**Important:** The replacement of the shared code piece occurs only within an array of the configuration JSON. In the above code, the shared code reference is `"script": ["{{ someOtherPlaceholder }}"]` which is the only valid form of a Shared Code reference. For example +`"script": ["some code {{ someOtherPlaceholder }} some other code"]` or `"script": "{{ someOtherPlaceholder }}"` are invalid Shared Code references which may not be replaced the way you intend. + +**Important:** The replacement of the shared code piece merges the `code_content` array containing the shared code definition with the array containing the shared code reference. With a shared code reference in form `"script": ["a", "{{ someOtherPlaceholder }}", "b"]` and shared code definition in form `"code_content": ["c", "d"]` the resulting replacement would be `"script": ["a", "c", "d", "b"]`. diff --git a/src/content/docs/integrate/variables/tutorial-1.png b/src/content/docs/integrate/variables/tutorial-1.png new file mode 100644 index 000000000..736929496 Binary files /dev/null and b/src/content/docs/integrate/variables/tutorial-1.png differ diff --git a/src/content/docs/integrate/variables/tutorial-2.png b/src/content/docs/integrate/variables/tutorial-2.png new file mode 100644 index 000000000..d0730e343 Binary files /dev/null and b/src/content/docs/integrate/variables/tutorial-2.png differ diff --git a/src/content/docs/integrate/variables/tutorial/index.md b/src/content/docs/integrate/variables/tutorial/index.md new file mode 100644 index 000000000..df655cacf --- /dev/null +++ b/src/content/docs/integrate/variables/tutorial/index.md @@ -0,0 +1,197 @@ +--- +title: Variables Tutorial +slug: 'integrate/variables/tutorial' +--- + + +This tutorial will guide you through basic usage of [Variables](/integrate/variables/) in the component configuration. +The result will be the parametrized configuration of the [Generic Extractor](https://developers.keboola.com/extend/generic-extractor), +but this approach can be applied to any component. + +In the examples, we use the `curl` console tool to interact with our APIs. + +## Define API endpoints + +First, store the [API endpoints](https://developers.keboola.com/overview/api/) as environment variables, so we don't have to repeat ourselves. + +We will need: +- [Storage API](/integrate/storage/api/) to store the variable definitions and the extractor configuration - +- [Job Queue API](https://developers.keboola.com/extend/job-queue/) to run the extractor job from the configuration. + +The host names depend on your [stack](https://developers.keboola.com/overview/api/#stacks-and-endpoints): + +```shell +export STORAGE_API_HOST="https://connection.keboola.com" +export JOB_QUEUE_HOST="https://queue.keboola.com" +``` + +## Obtain Storage API Token + +A [Storage API Token](/management/project/tokens/) is needed to interact with the [Keboola APIs](https://developers.keboola.com/overview/api/#list-of-keboola-apis). + +Obtain a Storage API token from the user interface of your project, see this [Guide](/management/project/tokens/). + +Then store the token to the environment variable. +```shell +export TOKEN="..." +``` + +## Define variables + +The next step is to define the variables in a [Variable Configuration](/integrate/variables/#variable-configuration). + +Define name and type of the variables. +```shell +export VARIABLE_CONFIG_NAME="Extractor variables" +export VARIABLE_CONFIG=' +{ + "variables": [ + { + "name": "outputBucket", + "type": "string" + }, + { + "name": "id", + "type": "int" + } + ] +} +' +``` + +Use [Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) to store *variable configuration*. +```shell +curl --include \ + --request POST \ + --header "Content-Type: application/x-www-form-urlencoded" \ + --header "X-StorageApi-Token: $TOKEN" \ + --data-urlencode "name=$VARIABLE_CONFIG_NAME" \ + --data-urlencode "configuration=$VARIABLE_CONFIG" \ +"$STORAGE_API_HOST/v2/storage/components/keboola.variables/configs" +``` + +Example API call result. +```json +{ + "id":"1234", + "name":"Extractor variables", + "description":"..." +} +``` + +Save *variable configuration* `id` from the the result to the environment variable. +```shell +export VARIABLE_CONFIG_ID="1234" +``` + +**The created *variable configuration* defines the names and types of variables.** + +You can create additional configurations that contain (default) [Variable Values](/integrate/variables/#variable-values). + +In this example, the values of the variables are entered directly to the [run API call](#run-extractor-configuration) (see bellow), +so configuration with the variable values is not used. + +## Create extractor configuration + +Define *extractor configuration* with variables `{{placeholders}}`. + +```shell + +export COMPONENT_ID="ex-generic-v2" +export EXTRACTOR_CONFIG_NAME="Extractor configuration" +export EXTRACTOR_CONFIG=' +{ + "parameters": { + "api": { + "baseUrl": "https://jsonplaceholder.typicode.com/" + }, + "config": { + "debug": true, + "outputBucket": "{{outputBucket}}", + "jobs": [ + { + "endpoint": "posts/{{id}}/comments" + } + ] + } + }, + "variables_id": "'$VARIABLE_CONFIG_ID'" +} +' + +``` + +Use [Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) to store extractor configuration. +```shell +curl --include \ + --request POST \ + --header "Content-Type: application/x-www-form-urlencoded" \ + --header "X-StorageApi-Token: $TOKEN" \ + --data-urlencode "name=$EXTRACTOR_CONFIG_NAME" \ + --data-urlencode "configuration=$EXTRACTOR_CONFIG" \ +"$STORAGE_API_HOST/v2/storage/components/$COMPONENT_ID/configs" +``` + +Example API call result. +```json +{ + "id":"4567", + "name":"Extractor configuration", + "description":"..." +} +``` + +Save *extractor configuration* `id` from the result to the environment variable. +```shell +export EXTRACTOR_CONFIG_ID="4567" +``` + +## Run extractor configuration + +Define values of the variables. +```shell +export VARIABLES_VALUES=' +[ + {"name": "outputBucket", "value": "my-bucket"}, + {"name": "id", "value": 1} +] +' +``` + +In this example are values of the variables part of the run job request. + +For other ways to define values see the [Variables documentation](/integrate/variables/#variable-values). + +Use [Run Job API call](https://api.keboola.com/?service=job-queue#post-/jobs) to run *extractor configuration*. +```shell +curl --include \ + --request POST \ + --header "Content-Type: application/json" \ + --header "X-StorageApi-Token: $TOKEN" \ + --data-binary ' + { + "component": "'$COMPONENT_ID'", + "config": "'$EXTRACTOR_CONFIG_ID'", + "mode": "run", + "variableValuesData": { + "values": '$VARIABLES_VALUES' + } + } + ' \ +"$JOB_QUEUE_HOST/jobs" +``` + +## Check the job result + +The status of a running job can be seen via API or UI. + +In the picture we can see that the entered values of the variables were used. + +![Screenshot -- Job](/integrate/variables/tutorial-1.png) + +A note about the replaced variables is in the job logs. + +![Screenshot -- Job Logs](/integrate/variables/tutorial-2.png) + +See the [Variables documentation](/integrate/variables/#variable-values) for more information. + diff --git a/src/content/docs/integrate/variables/variables.svg b/src/content/docs/integrate/variables/variables.svg new file mode 100644 index 000000000..5433c2a1c --- /dev/null +++ b/src/content/docs/integrate/variables/variables.svg @@ -0,0 +1,3 @@ + + +
variables_id
variables_id
variable_values_id
variable_values_id
Main Configuration
(vendor.component)
Main Configuration...
Variables Configuration
(keboola.variables)
Variables Configurat...
config
config
variableValuesId
variableValuesId
Orchestration
Orchestration
Variable Values Row
Variable Values...
config
config
variableValuesId
variableValuesId
Run Orchestration
Run O...
config
config
variableValuesId
variableValuesId
Run Configuration
Run C...
Viewer does not support full SVG 1.1
\ No newline at end of file diff --git a/src/content/docs/management/jobs/index.md b/src/content/docs/management/jobs/index.md index bdddd82d5..ef2462fb3 100644 --- a/src/content/docs/management/jobs/index.md +++ b/src/content/docs/management/jobs/index.md @@ -1,10 +1,12 @@ ---- -title: Jobs -slug: 'management/jobs' --- +title: Jobs +slug: 'management/jobs' +redirect_from: + - /integrate/jobs/ +--- + + - - Most of the things done in Keboola run as background, asynchronous jobs. For an overview of all jobs, running and finished, go to the **Jobs** section: diff --git a/src/content/docs/storage/files/index.md b/src/content/docs/storage/files/index.md index aeafc8c5a..829ee1371 100644 --- a/src/content/docs/storage/files/index.md +++ b/src/content/docs/storage/files/index.md @@ -66,7 +66,7 @@ Such a URL is valid for the entire validity of the file itself (either 15 days o In some cases, the file may be **sliced**. When you encounter a *sliced file*, you will obtain a [JSON](https://en.wikipedia.org/wiki/JSON) manifest file instead of the actual file. This can happen for some [exported or imported tables](/storage/tables/uploads/) from Storage or files which are particularly large. -Merging a sliced file requires a [substantial effort](https://developers.keboola.com/integrate/storage/api/import-export/#working-with-sliced-files). +Merging a sliced file requires a [substantial effort](/integrate/storage/api/import-export/#working-with-sliced-files). ## Limits The maximum allowed size of an uploaded file is currently 2 GB (2,048,000,000 bytes exactly). diff --git a/src/content/docs/storage/tables/uploads.md b/src/content/docs/storage/tables/uploads.md index 62c032d27..fe0ab349d 100644 --- a/src/content/docs/storage/tables/uploads.md +++ b/src/content/docs/storage/tables/uploads.md @@ -17,7 +17,7 @@ Every time a table is **exported** from Storage, the process is reversed: first, created in *Files* and then it is actually downloaded from there. This does not apply when exporting Storage tables manually though. Beware, however, that due to the nature of database exports, the exported table may be **sliced** and require -[substantial effort to reconstruct](https://developers.keboola.com/integrate/storage/api/import-export/#working-with-sliced-files). +[substantial effort to reconstruct](/integrate/storage/api/import-export/#working-with-sliced-files). To make sure your tables are exported as merged files, always use the **Export** feature in the **Action** tab of the table detail: diff --git a/src/content/docs/transformations/index.md b/src/content/docs/transformations/index.md index 269db4ae5..e843fc15d 100644 --- a/src/content/docs/transformations/index.md +++ b/src/content/docs/transformations/index.md @@ -193,7 +193,7 @@ Python and R transformations. Not available - API Interface + API Interface ✓ @@ -213,7 +213,7 @@ Python and R transformations. ### Transformations Transformations behave like any other [component](/components/). This means that they use the -standard [API](https://developers.keboola.com/integrate/storage/api/configurations/) to manipulate +standard [API](/integrate/storage/api/configurations/) to manipulate and run configurations and that creating your own [transformation components](https://developers.keboola.com/extend/component/) is possible. diff --git a/src/content/docs/transformations/variables/index.md b/src/content/docs/transformations/variables/index.md index 70f4537a2..f499299e9 100644 --- a/src/content/docs/transformations/variables/index.md +++ b/src/content/docs/transformations/variables/index.md @@ -10,6 +10,10 @@ which differ in only a limited number of values. You can have, for example, a tr processes all orders from the Meals department. With variables, you can modify it to work for the Drinks department, too. +:::tip +Want to define and resolve variables programmatically? See [Variables](/integrate/variables/) for the underlying Configuration API used by the `keboola.variables` component. +::: + ## Variables Transformation variables are unrelated to the transformation code itself. It means that they do not manifest themselves as SQL or Python variables. Transformation variables are evaluated before the transformation is run and diff --git a/src/content/docs/workspace/table-export.md b/src/content/docs/workspace/table-export.md index e589e6d1f..3a74b60d3 100644 --- a/src/content/docs/workspace/table-export.md +++ b/src/content/docs/workspace/table-export.md @@ -55,7 +55,7 @@ finishes, its `results` contain the ID of the exported file: } ``` -Download the file with the standard [file download](/integrate/storage/api/importer/#download-a-file) flow. +Download the file with the standard [file download](/integrate/storage/api/import-export/#working-with-data) flow. ## Backend-Specific Notes diff --git a/src/sidebar.mjs b/src/sidebar.mjs index 86a768d09..9e0810832 100644 --- a/src/sidebar.mjs +++ b/src/sidebar.mjs @@ -529,6 +529,44 @@ export const sidebar = [ { slug: "automate/run-job" }, { slug: "automate/run-orchestration" }, { slug: "automate/set-schedule" }, + { + label: "Integration", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate" }, + { + label: "Storage API", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate/storage/api" }, + { slug: "integrate/storage/api/configurations" }, + { slug: "integrate/storage/api/import-export" }, + { slug: "integrate/storage/api/importer" }, + { slug: "integrate/storage/api/tde-exporter" }, + ], + }, + { slug: "integrate/storage/python-client" }, + { slug: "integrate/storage/r-client" }, + { slug: "integrate/storage/php-client" }, + { slug: "integrate/storage/docker-cli-client" }, + { + label: "Variables", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate/variables" }, + { slug: "integrate/variables/tutorial" }, + ], + }, + { + label: "Artifacts", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate/artifacts" }, + { slug: "integrate/artifacts/tutorial" }, + ], + }, + ], + }, ], }, ];