diff --git a/MIGRATION-REPORT.md b/MIGRATION-REPORT.md new file mode 100644 index 000000000..8fd0b86eb --- /dev/null +++ b/MIGRATION-REPORT.md @@ -0,0 +1,216 @@ +# MIGRATION REPORT — developers.keboola.com → help (phase 1, scripted) + +Generated by `scripts/migrate-devdocs.mjs` — re-run it to reproduce this tree byte-for-byte. + +- pages ported: **165** · skipped: **10** · images: **200** · assets: **4** +- includes inlined: 9 · relative image refs fixed: 1 · inbound links flipped: 0 +- redirects injected: 3 · banners: cli/keboola-as-code · nav: Developer Docs group mirrors dev navigation.yml (161 entries) + +## Skipped (woven by open weave PRs / dead chrome) (10) + +- 404.md (/404.html — not migrated) +- index.md (/ — not migrated) +- integrate/data-streams/index.md (/integrate/data-streams/ — canonical /storage/data-streams/) +- integrate/data-streams/overview/index.md (/integrate/data-streams/overview/ — canonical /storage/data-streams/) +- integrate/data-streams/tutorial/index.md (/integrate/data-streams/tutorial/ — canonical /storage/data-streams/) +- integrate/database/index.md (/integrate/database/ — canonical /components/extractors/database/) +- integrate/mcp.md (/integrate/mcp/ — canonical /ai/mcp-server/) +- integrate/orchestrator/index.md (/integrate/orchestrator/ — canonical /flows/) +- overview/index.md (/overview/ — canonical /overview/) +- overview/repositories.md (/overview/repositories/ — canonical /overview/) + +## Redirects injected into main canonicals (3) + +- /integrate/mcp/ (already present on ai/mcp-server/index.md) +- /integrate/database/ (already present on components/extractors/database/index.md) +- /integrate/orchestrator/ (already present on flows/index.md) + +## Includes inlined (9) + +- extend/common-interface/development-branches.md: branches-beta-warning.html -> admonition +- extend/generic-extractor/configuration/configuration.md: branches-beta-warning.html -> admonition +- extend/generic-extractor/configuration/configuration.md: config-map.json inlined (149 lines) +- extend/generic-extractor/configuration/configuration.md: config-events.js inlined (67 lines) +- extend/generic-extractor/map.md: config-map.json inlined (149 lines) +- extend/generic-extractor/map.md: config-events.js inlined (67 lines) +- extend/generic-writer/configuration/configuration.md: writer-config-map.json inlined (84 lines) +- extend/generic-writer/configuration/configuration.md: writer-config-events.js inlined (40 lines) +- integrate/storage/api/import-export.md: async-create.py inlined (74 lines) + +## Relative image refs repaired (1) + +- extend/generic-extractor/tutorial/jobs.md: child_debug.png -> /extend/generic-extractor/tutorial/child_debug.png + +## Inbound links flipped (outside ported tree) (0) + + +## TODO(human-review) (0) + + +## Ported pages (165) + +- /automate/ -> /automate/ +- /automate/run-job/ -> /automate/run-job/ +- /automate/run-orchestration/ -> /automate/run-orchestration/ +- /automate/set-schedule/ -> /automate/set-schedule/ +- /cli/commands/ci/ -> /cli/keboola-as-code/commands/ci/ +- /cli/commands/ci/workflows/ -> /cli/keboola-as-code/commands/ci/workflows/ +- /cli/commands/dbt/generate/env/ -> /cli/keboola-as-code/commands/dbt/generate/env/ +- /cli/commands/dbt/generate/ -> /cli/keboola-as-code/commands/dbt/generate/ +- /cli/commands/dbt/generate/profile/ -> /cli/keboola-as-code/commands/dbt/generate/profile/ +- /cli/commands/dbt/generate/sources/ -> /cli/keboola-as-code/commands/dbt/generate/sources/ +- /cli/commands/dbt/ -> /cli/keboola-as-code/commands/dbt/ +- /cli/commands/dbt/init/ -> /cli/keboola-as-code/commands/dbt/init/ +- /cli/commands/help/ -> /cli/keboola-as-code/commands/help/ +- /cli/commands/ -> /cli/keboola-as-code/commands/ +- /cli/commands/llm/export/ -> /cli/keboola-as-code/commands/llm/export/ +- /cli/commands/llm/ -> /cli/keboola-as-code/commands/llm/ +- /cli/commands/llm/init/ -> /cli/keboola-as-code/commands/llm/init/ +- /cli/commands/local/create/config/ -> /cli/keboola-as-code/commands/local/create/config/ +- /cli/commands/local/create/ -> /cli/keboola-as-code/commands/local/create/ +- /cli/commands/local/create/row/ -> /cli/keboola-as-code/commands/local/create/row/ +- /cli/commands/local/encrypt/ -> /cli/keboola-as-code/commands/local/encrypt/ +- /cli/commands/local/fix-paths/ -> /cli/keboola-as-code/commands/local/fix-paths/ +- /cli/commands/local/ -> /cli/keboola-as-code/commands/local/ +- /cli/commands/local/persist/ -> /cli/keboola-as-code/commands/local/persist/ +- /cli/commands/local/validate/config/ -> /cli/keboola-as-code/commands/local/validate/config/ +- /cli/commands/local/validate/ -> /cli/keboola-as-code/commands/local/validate/ +- /cli/commands/local/validate/row/ -> /cli/keboola-as-code/commands/local/validate/row/ +- /cli/commands/local/validate/schema/ -> /cli/keboola-as-code/commands/local/validate/schema/ +- /cli/commands/remote/create/branch/ -> /cli/keboola-as-code/commands/remote/create/branch/ +- /cli/commands/remote/create/bucket/ -> /cli/keboola-as-code/commands/remote/create/bucket/ +- /cli/commands/remote/create/ -> /cli/keboola-as-code/commands/remote/create/ +- /cli/commands/remote/file/download/ -> /cli/keboola-as-code/commands/remote/file/download/ +- /cli/commands/remote/file/ -> /cli/keboola-as-code/commands/remote/file/ +- /cli/commands/remote/file/upload/ -> /cli/keboola-as-code/commands/remote/file/upload/ +- /cli/commands/remote/ -> /cli/keboola-as-code/commands/remote/ +- /cli/commands/remote/job/ -> /cli/keboola-as-code/commands/remote/job/ +- /cli/commands/remote/job/run/ -> /cli/keboola-as-code/commands/remote/job/run/ +- /cli/commands/remote/table/create/ -> /cli/keboola-as-code/commands/remote/table/create/ +- /cli/commands/remote/table/detail/ -> /cli/keboola-as-code/commands/remote/table/detail/ +- /cli/commands/remote/table/download/ -> /cli/keboola-as-code/commands/remote/table/download/ +- /cli/commands/remote/table/import/ -> /cli/keboola-as-code/commands/remote/table/import/ +- /cli/commands/remote/table/ -> /cli/keboola-as-code/commands/remote/table/ +- /cli/commands/remote/table/preview/ -> /cli/keboola-as-code/commands/remote/table/preview/ +- /cli/commands/remote/table/unload/ -> /cli/keboola-as-code/commands/remote/table/unload/ +- /cli/commands/remote/table/upload/ -> /cli/keboola-as-code/commands/remote/table/upload/ +- /cli/commands/remote/workspace/create/ -> /cli/keboola-as-code/commands/remote/workspace/create/ +- /cli/commands/remote/workspace/delete/ -> /cli/keboola-as-code/commands/remote/workspace/delete/ +- /cli/commands/remote/workspace/detail/ -> /cli/keboola-as-code/commands/remote/workspace/detail/ +- /cli/commands/remote/workspace/ -> /cli/keboola-as-code/commands/remote/workspace/ +- /cli/commands/remote/workspace/list/ -> /cli/keboola-as-code/commands/remote/workspace/list/ +- /cli/commands/status/ -> /cli/keboola-as-code/commands/status/ +- /cli/commands/sync/diff/ -> /cli/keboola-as-code/commands/sync/diff/ +- /cli/commands/sync/ -> /cli/keboola-as-code/commands/sync/ +- /cli/commands/sync/init/ -> /cli/keboola-as-code/commands/sync/init/ +- /cli/commands/sync/pull/ -> /cli/keboola-as-code/commands/sync/pull/ +- /cli/commands/sync/push/ -> /cli/keboola-as-code/commands/sync/push/ +- /cli/dbt/ -> /cli/keboola-as-code/dbt/ +- /cli/devops-use-cases/ -> /cli/keboola-as-code/devops-use-cases/ +- /cli/getting-started/ -> /cli/keboola-as-code/getting-started/ +- /cli/github-integration/ -> /cli/keboola-as-code/github-integration/ +- /cli/ -> /cli/keboola-as-code/ +- /cli/installation/ -> /cli/keboola-as-code/installation/ +- /cli/structure/ -> /cli/keboola-as-code/structure/ +- /extend/common-interface/actions/ -> /extend/common-interface/actions/ +- /extend/common-interface/config-file/ -> /extend/common-interface/config-file/ +- /extend/common-interface/development-branches/ -> /extend/common-interface/development-branches/ +- /extend/common-interface/environment/ -> /extend/common-interface/environment/ +- /extend/common-interface/folders/ -> /extend/common-interface/folders/ +- /extend/common-interface/ -> /extend/common-interface/ +- /extend/common-interface/logging/ -> /extend/common-interface/logging/ +- /extend/common-interface/manifest-files/in-files-abs-staging/ -> /extend/common-interface/manifest-files/in-files-abs-staging/ +- /extend/common-interface/manifest-files/in-files-manifests/ -> /extend/common-interface/manifest-files/in-files-manifests/ +- /extend/common-interface/manifest-files/in-files-s3-staging/ -> /extend/common-interface/manifest-files/in-files-s3-staging/ +- /extend/common-interface/manifest-files/in-tables-manifests/ -> /extend/common-interface/manifest-files/in-tables-manifests/ +- /extend/common-interface/manifest-files/out-files-manifests/ -> /extend/common-interface/manifest-files/out-files-manifests/ +- /extend/common-interface/manifest-files/out-tables-manifests-native-types/ -> /extend/common-interface/manifest-files/out-tables-manifests-native-types/ +- /extend/common-interface/manifest-files/out-tables-manifests/ -> /extend/common-interface/manifest-files/out-tables-manifests/ +- /extend/common-interface/manifest-files/ -> /extend/common-interface/manifest-files/ +- /extend/common-interface/oauth/ -> /extend/common-interface/oauth/ +- /extend/component/code-patterns/ -> /extend/component/code-patterns/ +- /extend/component/code-patterns/interface/ -> /extend/component/code-patterns/interface/ +- /extend/component/code-patterns/tutorial/ -> /extend/component/code-patterns/tutorial/ +- /extend/component/deployment/ -> /extend/component/deployment/ +- /extend/component/implementation/ -> /extend/component/implementation/ +- /extend/component/implementation/php/ -> /extend/component/implementation/php/ +- /extend/component/implementation/python/ -> /extend/component/implementation/python/ +- /extend/component/implementation/r/ -> /extend/component/implementation/r/ +- /extend/component/ -> /extend/component/ +- /extend/component/processors/ -> /extend/component/processors/ +- /extend/component/running/ -> /extend/component/running/ +- /extend/component/tutorial/configuration/ -> /extend/component/tutorial/configuration/ +- /extend/component/tutorial/debugging/ -> /extend/component/tutorial/debugging/ +- /extend/component/tutorial/ -> /extend/component/tutorial/ +- /extend/component/tutorial/input-mapping/ -> /extend/component/tutorial/input-mapping/ +- /extend/component/tutorial/output-mapping/ -> /extend/component/tutorial/output-mapping/ +- /extend/component/tutorial/processors/ -> /extend/component/tutorial/processors/ +- /extend/component/ui-options/configuration-schema/ -> /extend/component/ui-options/configuration-schema/ +- /extend/component/ui-options/default-configuration/ -> /extend/component/ui-options/default-configuration/ +- /extend/component/ui-options/ -> /extend/component/ui-options/ +- /extend/component/ui-options/configuration-schema/examples/ -> /extend/component/ui-options/configuration-schema/examples/ +- /extend/component/ui-options/configuration-schema/sync-action-examples/ -> /extend/component/ui-options/configuration-schema/sync-action-examples/ +- /extend/generic-extractor/configuration/api/authentication/api_key/ -> /extend/generic-extractor/configuration/api/authentication/api_key/ +- /extend/generic-extractor/configuration/api/authentication/basic/ -> /extend/generic-extractor/configuration/api/authentication/basic/ +- /extend/generic-extractor/configuration/api/authentication/bearer_token/ -> /extend/generic-extractor/configuration/api/authentication/bearer_token/ +- /extend/generic-extractor/configuration/api/authentication/ -> /extend/generic-extractor/configuration/api/authentication/ +- /extend/generic-extractor/configuration/api/authentication/login/ -> /extend/generic-extractor/configuration/api/authentication/login/ +- /extend/generic-extractor/configuration/api/authentication/oauth10/ -> /extend/generic-extractor/configuration/api/authentication/oauth10/ +- /extend/generic-extractor/configuration/api/authentication/oauth20-login/ -> /extend/generic-extractor/configuration/api/authentication/oauth20-login/ +- /extend/generic-extractor/configuration/api/authentication/oauth20/ -> /extend/generic-extractor/configuration/api/authentication/oauth20/ +- /extend/generic-extractor/configuration/api/authentication/oauth_cc/ -> /extend/generic-extractor/configuration/api/authentication/oauth_cc/ +- /extend/generic-extractor/configuration/api/authentication/query/ -> /extend/generic-extractor/configuration/api/authentication/query/ +- /extend/generic-extractor/configuration/api/ -> /extend/generic-extractor/configuration/api/ +- /extend/generic-extractor/configuration/api/pagination/cursor/ -> /extend/generic-extractor/configuration/api/pagination/cursor/ +- /extend/generic-extractor/configuration/api/pagination/ -> /extend/generic-extractor/configuration/api/pagination/ +- /extend/generic-extractor/configuration/api/pagination/multiple/ -> /extend/generic-extractor/configuration/api/pagination/multiple/ +- /extend/generic-extractor/configuration/api/pagination/offset/ -> /extend/generic-extractor/configuration/api/pagination/offset/ +- /extend/generic-extractor/configuration/api/pagination/pagenum/ -> /extend/generic-extractor/configuration/api/pagination/pagenum/ +- /extend/generic-extractor/configuration/api/pagination/response-param/ -> /extend/generic-extractor/configuration/api/pagination/response-param/ +- /extend/generic-extractor/configuration/api/pagination/response-url/ -> /extend/generic-extractor/configuration/api/pagination/response-url/ +- /extend/generic-extractor/configuration/aws-signature/ -> /extend/generic-extractor/configuration/aws-signature/ +- /extend/generic-extractor/configuration/config/ -> /extend/generic-extractor/configuration/config/ +- /extend/generic-extractor/configuration/config/jobs/children/ -> /extend/generic-extractor/configuration/config/jobs/children/ +- /extend/generic-extractor/configuration/config/jobs/ -> /extend/generic-extractor/configuration/config/jobs/ +- /extend/generic-extractor/configuration/config/mappings/ -> /extend/generic-extractor/configuration/config/mappings/ +- /extend/generic-extractor/configuration/ -> /extend/generic-extractor/configuration/ +- /extend/generic-extractor/configuration/iterations/ -> /extend/generic-extractor/configuration/iterations/ +- /extend/generic-extractor/configuration/ssh-proxy/ -> /extend/generic-extractor/configuration/ssh-proxy/ +- /extend/generic-extractor/functions/ -> /extend/generic-extractor/functions/ +- /extend/generic-extractor/incremental/ -> /extend/generic-extractor/incremental/ +- /extend/generic-extractor/ -> /extend/generic-extractor/ +- /extend/generic-extractor/map/ -> /extend/generic-extractor/map/ +- /extend/generic-extractor/publish/ -> /extend/generic-extractor/publish/ +- /extend/generic-extractor/running/ -> /extend/generic-extractor/running/ +- /extend/generic-extractor/tutorial/basic/ -> /extend/generic-extractor/tutorial/basic/ +- /extend/generic-extractor/tutorial/ -> /extend/generic-extractor/tutorial/ +- /extend/generic-extractor/tutorial/jobs/ -> /extend/generic-extractor/tutorial/jobs/ +- /extend/generic-extractor/tutorial/json/ -> /extend/generic-extractor/tutorial/json/ +- /extend/generic-extractor/tutorial/mapping/ -> /extend/generic-extractor/tutorial/mapping/ +- /extend/generic-extractor/tutorial/pagination/ -> /extend/generic-extractor/tutorial/pagination/ +- /extend/generic-extractor/tutorial/rest/ -> /extend/generic-extractor/tutorial/rest/ +- /extend/generic-writer/configuration-examples/ -> /extend/generic-writer/configuration-examples/ +- /extend/generic-writer/configuration/ -> /extend/generic-writer/configuration/ +- /extend/generic-writer/ -> /extend/generic-writer/ +- /extend/ -> /extend/ +- /extend/job-queue/ -> /extend/job-queue/ +- /extend/publish/checklist/ -> /extend/publish/checklist/ +- /extend/publish/ -> /extend/publish/ +- /integrate/artifacts/ -> /integrate/artifacts/ +- /integrate/artifacts/tutorial/ -> /integrate/artifacts/tutorial/ +- /integrate/ -> /integrate/ +- /integrate/jobs/ -> /integrate/jobs/ +- /integrate/storage/api/configurations/ -> /integrate/storage/api/configurations/ +- /integrate/storage/api/import-export/ -> /integrate/storage/api/import-export/ +- /integrate/storage/api/importer/ -> /integrate/storage/api/importer/ +- /integrate/storage/api/ -> /integrate/storage/api/ +- /integrate/storage/api/tde-exporter/ -> /integrate/storage/api/tde-exporter/ +- /integrate/storage/docker-cli-client/ -> /integrate/storage/docker-cli-client/ +- /integrate/storage/ -> /integrate/storage/ +- /integrate/storage/php-client/ -> /integrate/storage/php-client/ +- /integrate/storage/python-client/ -> /integrate/storage/python-client/ +- /integrate/storage/r-client/ -> /integrate/storage/r-client/ +- /integrate/variables/ -> /integrate/variables/ +- /integrate/variables/tutorial/ -> /integrate/variables/tutorial/ +- /overview/api/ -> /overview/api/ +- /overview/encryption/ -> /overview/encryption/ diff --git a/_data/navigation.yml b/_data/navigation.yml index d448777da..8b243dab1 100644 --- a/_data/navigation.yml +++ b/_data/navigation.yml @@ -721,6 +721,375 @@ items: - url: /ai/mcp-server/ title: MCP Server + + # --- Developer Docs (migrated from developers.keboola.com, phase 1) --- + - url: /overview/api/ + title: Developer Docs + items: + - url: /overview/api/ + title: Our APIs + - url: /overview/encryption/ + title: Encryption + - url: /extend/ + title: Extending Keboola + items: + - url: /extend/component/ + title: Components + items: + - url: /extend/component/tutorial/ + title: Tutorial + items: + - url: /extend/component/tutorial/input-mapping/ + title: Input Mapping + - url: /extend/component/tutorial/output-mapping/ + title: Output Mapping + - url: /extend/component/tutorial/configuration/ + title: Configuration + - url: /extend/component/tutorial/processors/ + title: Processors + - url: /extend/component/tutorial/debugging/ + title: Debugging + - url: /extend/component/processors/ + title: Processors + - url: /extend/component/code-patterns/ + title: Code Patterns + items: + - url: /extend/component/code-patterns/interface/ + title: Interface + - url: /extend/component/code-patterns/tutorial/ + title: Tutorial + - url: /extend/component/implementation/ + title: Implementation Notes + items: + - url: /extend/component/implementation/php/ + title: PHP Implementation Notes + - url: /extend/component/implementation/python/ + title: Python Implementation Notes + - url: /extend/component/implementation/r/ + title: R Implementation Notes + - url: /extend/component/running/ + title: Running Components + - url: /extend/component/ui-options/ + title: UI Options + items: + - url: /extend/component/ui-options/configuration-schema/ + title: Configuration Schema + items: + - url: /extend/component/ui-options/configuration-schema/examples + title: Examples + - url: /extend/component/ui-options/configuration-schema/sync-action-examples + title: Sync Action Examples + - url: /extend/component/ui-options/default-configuration/ + title: Default Configuration + - url: /extend/component/deployment/ + title: Deployment + - url: /extend/generic-extractor/ + title: Generic Extractor + items: + - url: /extend/generic-extractor/tutorial/ + title: Generic Extractor Tutorial + items: + - url: /extend/generic-extractor/tutorial/rest/ + title: REST HTTP API Introduction + - url: /extend/generic-extractor/tutorial/json/ + title: JSON Introduction + - url: /extend/generic-extractor/tutorial/basic/ + title: Basic Configuration + - url: /extend/generic-extractor/tutorial/pagination/ + title: Pagination Tutorial + - url: /extend/generic-extractor/tutorial/jobs/ + title: Jobs Tutorial + - url: /extend/generic-extractor/tutorial/mapping/ + title: Mapping Tutorial + - url: /extend/generic-extractor/configuration/ + title: Configuration + items: + - url: /extend/generic-extractor/configuration/api/ + title: API Configuration + items: + - url: /extend/generic-extractor/configuration/api/pagination/ + title: Pagination + items: + - url: /extend/generic-extractor/configuration/api/pagination/response-url/ + title: Response URL Scroller + - url: /extend/generic-extractor/configuration/api/pagination/response-param/ + title: Response Parameter Scroller + - url: /extend/generic-extractor/configuration/api/pagination/offset/ + title: Offset Scroller + - url: /extend/generic-extractor/configuration/api/pagination/pagenum/ + title: Page Number Scroller + - url: /extend/generic-extractor/configuration/api/pagination/cursor/ + title: Cursor Scroller + - url: /extend/generic-extractor/configuration/api/pagination/multiple/ + title: Multiple Scrollers + - url: /extend/generic-extractor/configuration/api/authentication/ + title: Authentication + items: + - url: /extend/generic-extractor/configuration/api/authentication/query/ + title: Query + - url: /extend/generic-extractor/configuration/api/authentication/basic/ + title: Basic + - url: /extend/generic-extractor/configuration/api/authentication/bearer_token/ + title: Bearer Token + - url: /extend/generic-extractor/configuration/api/authentication/api_key/ + title: API Key + - url: /extend/generic-extractor/configuration/api/authentication/login/ + title: Login + - url: /extend/generic-extractor/configuration/api/authentication/oauth_cc/ + title: OAuth 2.0 Client Credentials + - url: /extend/generic-extractor/configuration/api/authentication/oauth10/ + title: OAuth 1.0 + - url: /extend/generic-extractor/configuration/api/authentication/oauth20/ + title: OAuth 2.0 + - url: /extend/generic-extractor/configuration/api/authentication/oauth20-login/ + title: Login using OAuth 2.0 + - url: /extend/generic-extractor/configuration/config/ + title: Extraction Configuration + items: + - url: /extend/generic-extractor/configuration/config/jobs/ + title: Jobs + items: + - url: /extend/generic-extractor/configuration/config/jobs/children/ + title: Child Jobs + - url: /extend/generic-extractor/configuration/config/mappings/ + title: Mappings + - url: /extend/generic-extractor/configuration/iterations/ + title: Iterations + - url: /extend/generic-extractor/configuration/ssh-proxy/ + title: SSH Proxy Configuration + - url: /extend/generic-extractor/map/ + title: Configuration Map + - url: /extend/generic-extractor/functions/ + title: Functions + - url: /extend/generic-extractor/incremental/ + title: Incremental Extraction + - url: /extend/generic-extractor/running/ + title: Running Generic Extractor + - url: /extend/generic-extractor/publish/ + title: Publishing Component + - url: /extend/generic-writer/ + title: Generic Writer + items: + - url: /extend/generic-writer/configuration/ + title: Configuration + - url: /extend/generic-writer/configuration-examples/ + title: Configuration Examples + - url: /extend/common-interface/ + title: Common Interface + items: + - url: /extend/common-interface/folders/ + title: Data Folders + - url: /extend/common-interface/config-file/ + title: Configuration File + - url: /extend/common-interface/environment/ + title: Environment + - url: /extend/common-interface/manifest-files/ + title: Manifest Files + items: + - url: /extend/common-interface/manifest-files/in-tables-manifests/ + title: IN tables + - url: /extend/common-interface/manifest-files/in-files-manifests/ + title: IN files + - url: /extend/common-interface/manifest-files/in-files-s3-staging/ + title: IN files S3 staging + - url: /extend/common-interface/manifest-files/in-files-abs-staging/ + title: IN files ABS staging + - url: /extend/common-interface/manifest-files/out-tables-manifests/ + title: OUT tables + - url: /extend/common-interface/manifest-files/out-tables-manifests-native-types/ + title: OUT tables with Native Types + - url: /extend/common-interface/manifest-files/out-files-manifests/ + title: OUT files + - url: /extend/common-interface/oauth/ + title: OAuth2 + - url: /extend/common-interface/actions/ + title: Actions + - url: /extend/common-interface/logging/ + title: Logging + - url: /extend/common-interface/development-branches/ + title: Development branches + - url: /extend/job-queue/ + title: Job Queue + - url: /extend/publish/ + title: Publishing Component + items: + - url: /extend/publish/checklist/ + title: Checklist + - url: /integrate/ + title: Integration + items: + - url: /integrate/storage/ + title: Storage + items: + - url: /integrate/storage/php-client/ + title: PHP client library + - url: /integrate/storage/r-client/ + title: R client library + - url: /integrate/storage/python-client/ + title: Python client library + - url: /integrate/storage/docker-cli-client/ + title: Docker CLI client + - url: /integrate/storage/api/ + title: Using API + items: + - url: /integrate/storage/api/configurations/ + title: Configurations API + - url: /integrate/storage/api/importer/ + title: Storage API Importer + - url: /integrate/storage/api/import-export/ + title: Manually importing and exporting data + - url: /integrate/storage/api/tde-exporter/ + title: TDE Exporter + - url: /integrate/jobs/ + title: Component Jobs + - url: /integrate/variables/ + title: Variables + items: + - url: /integrate/variables/tutorial/ + title: Tutorial + - url: /integrate/artifacts/ + title: Artifacts + items: + - url: /integrate/artifacts/tutorial/ + title: Tutorial + - url: /automate/ + title: Automation/Common Tasks + items: + - url: /automate/run-job/ + title: Run Job + - url: /automate/run-orchestration/ + title: Run Orchestration + - url: /automate/set-schedule/ + title: Set Schedule + - url: /cli/keboola-as-code/ + title: Keboola as Code CLI + items: + - url: /cli/keboola-as-code/installation/ + title: Installation + - url: /cli/keboola-as-code/getting-started/ + title: Getting Started + - url: /cli/keboola-as-code/structure/ + title: Structure + - url: /cli/keboola-as-code/commands/ + title: Commands + items: + - url: /cli/keboola-as-code/commands/help/ + title: help + - url: /cli/keboola-as-code/commands/status/ + title: status + - url: /cli/keboola-as-code/commands/sync/ + title: sync + items: + - url: /cli/keboola-as-code/commands/sync/init/ + title: init + - url: /cli/keboola-as-code/commands/sync/pull/ + title: pull + - url: /cli/keboola-as-code/commands/sync/push/ + title: push + - url: /cli/keboola-as-code/commands/sync/diff/ + title: diff + - url: /cli/keboola-as-code/commands/ci/ + title: ci + items: + - url: /cli/keboola-as-code/commands/ci/workflows/ + title: workflows + - url: /cli/keboola-as-code/commands/local/ + title: local + items: + - url: /cli/keboola-as-code/commands/local/create/ + title: create + items: + - url: /cli/keboola-as-code/commands/local/create/config/ + title: config + - url: /cli/keboola-as-code/commands/local/create/row/ + title: row + - url: /cli/keboola-as-code/commands/local/persist/ + title: persist + - url: /cli/keboola-as-code/commands/local/encrypt/ + title: encrypt + - url: /cli/keboola-as-code/commands/local/validate/ + title: validate + items: + - url: /cli/keboola-as-code/commands/local/validate/config/ + title: config + - url: /cli/keboola-as-code/commands/local/validate/row/ + title: row + - url: /cli/keboola-as-code/commands/local/validate/schema/ + title: schema + - url: /cli/keboola-as-code/commands/local/fix-paths/ + title: fix-paths + - url: /cli/keboola-as-code/commands/remote/ + title: remote + items: + - url: /cli/keboola-as-code/commands/remote/create/ + title: create + items: + - url: /cli/keboola-as-code/commands/remote/create/branch/ + title: branch + - url: /cli/keboola-as-code/commands/remote/create/bucket/ + title: bucket + - url: /cli/keboola-as-code/commands/remote/file/ + title: file + items: + - url: /cli/keboola-as-code/commands/remote/file/download/ + title: download + - url: /cli/keboola-as-code/commands/remote/file/upload/ + title: upload + - url: /cli/keboola-as-code/commands/remote/job/ + title: job + items: + - url: /cli/keboola-as-code/commands/remote/job/run/ + title: run + - url: /cli/keboola-as-code/commands/remote/table/ + title: table + items: + - url: /cli/keboola-as-code/commands/remote/table/create/ + title: create + - url: /cli/keboola-as-code/commands/remote/table/upload/ + title: upload + - url: /cli/keboola-as-code/commands/remote/table/download/ + title: download + - url: /cli/keboola-as-code/commands/remote/table/preview/ + title: preview + - url: /cli/keboola-as-code/commands/remote/table/detail/ + title: detail + - url: /cli/keboola-as-code/commands/remote/table/import/ + title: import + - url: /cli/keboola-as-code/commands/remote/table/unload/ + title: unload + - url: /cli/keboola-as-code/commands/remote/workspace/ + title: workspace + items: + - url: /cli/keboola-as-code/commands/remote/workspace/create/ + title: create + - url: /cli/keboola-as-code/commands/remote/workspace/delete/ + title: delete + - url: /cli/keboola-as-code/commands/remote/workspace/detail/ + title: detail + - url: /cli/keboola-as-code/commands/remote/workspace/list/ + title: list + - url: /cli/keboola-as-code/commands/dbt/ + title: dbt + items: + - url: /cli/keboola-as-code/commands/dbt/init/ + title: init + - url: /cli/keboola-as-code/commands/dbt/generate/ + title: generate + items: + - url: /cli/keboola-as-code/commands/dbt/generate/profile/ + title: profile + - url: /cli/keboola-as-code/commands/dbt/generate/sources/ + title: sources + - url: /cli/keboola-as-code/commands/dbt/generate/env/ + title: env + - url: /cli/keboola-as-code/github-integration/ + title: GitHub Integration + - url: /cli/keboola-as-code/devops-use-cases/ + title: DevOps Use Cases + - url: /cli/keboola-as-code/dbt/ + title: dbt + + - url: /external-integrations/ title: External Integrations items: diff --git a/public/automate/job-parameters.png b/public/automate/job-parameters.png new file mode 100644 index 000000000..2e8135532 Binary files /dev/null and b/public/automate/job-parameters.png differ diff --git a/public/automate/job-row-parameters.png b/public/automate/job-row-parameters.png new file mode 100644 index 000000000..d48d7ec39 Binary files /dev/null and b/public/automate/job-row-parameters.png differ diff --git a/public/automate/orchestration-parameters.png b/public/automate/orchestration-parameters.png new file mode 100644 index 000000000..83fafe0a4 Binary files /dev/null and b/public/automate/orchestration-parameters.png differ diff --git a/public/automate/token-settings.png b/public/automate/token-settings.png new file mode 100644 index 000000000..0266b27f7 Binary files /dev/null and b/public/automate/token-settings.png differ diff --git a/public/cli/keboola-as-code/devops-use-cases/branch_management.png b/public/cli/keboola-as-code/devops-use-cases/branch_management.png new file mode 100644 index 000000000..0bca08692 Binary files /dev/null and b/public/cli/keboola-as-code/devops-use-cases/branch_management.png differ diff --git a/public/cli/keboola-as-code/devops-use-cases/dev_prod_flow.png b/public/cli/keboola-as-code/devops-use-cases/dev_prod_flow.png new file mode 100644 index 000000000..1182aba69 Binary files /dev/null and b/public/cli/keboola-as-code/devops-use-cases/dev_prod_flow.png differ diff --git a/public/cli/keboola-as-code/devops-use-cases/devprod.png b/public/cli/keboola-as-code/devops-use-cases/devprod.png new file mode 100644 index 000000000..b620a2b1d Binary files /dev/null and b/public/cli/keboola-as-code/devops-use-cases/devprod.png differ diff --git a/public/cli/keboola-as-code/devops-use-cases/devtools.png b/public/cli/keboola-as-code/devops-use-cases/devtools.png new file mode 100644 index 000000000..5b23a720b Binary files /dev/null and b/public/cli/keboola-as-code/devops-use-cases/devtools.png differ diff --git a/public/cli/keboola-as-code/devops-use-cases/init.png b/public/cli/keboola-as-code/devops-use-cases/init.png new file mode 100644 index 000000000..7c23263a1 Binary files /dev/null and b/public/cli/keboola-as-code/devops-use-cases/init.png differ diff --git a/public/cli/keboola-as-code/devops-use-cases/project_deploy.png b/public/cli/keboola-as-code/devops-use-cases/project_deploy.png new file mode 100644 index 000000000..c8db9616a Binary files /dev/null and b/public/cli/keboola-as-code/devops-use-cases/project_deploy.png differ diff --git a/public/cli/keboola-as-code/getting-started/configurations-copy-1.jpg b/public/cli/keboola-as-code/getting-started/configurations-copy-1.jpg new file mode 100644 index 000000000..83d58e23f Binary files /dev/null and b/public/cli/keboola-as-code/getting-started/configurations-copy-1.jpg differ diff --git a/public/cli/keboola-as-code/getting-started/configurations-copy-2.jpg b/public/cli/keboola-as-code/getting-started/configurations-copy-2.jpg new file mode 100644 index 000000000..7cfc027f0 Binary files /dev/null and b/public/cli/keboola-as-code/getting-started/configurations-copy-2.jpg differ diff --git a/public/cli/keboola-as-code/github-integration/github-actions.jpg b/public/cli/keboola-as-code/github-integration/github-actions.jpg new file mode 100644 index 000000000..70f380f6c Binary files /dev/null and b/public/cli/keboola-as-code/github-integration/github-actions.jpg differ diff --git a/public/cli/keboola-as-code/github-integration/pull-commit.jpg b/public/cli/keboola-as-code/github-integration/pull-commit.jpg new file mode 100644 index 000000000..7c4129000 Binary files /dev/null and b/public/cli/keboola-as-code/github-integration/pull-commit.jpg differ diff --git a/public/cli/keboola-as-code/github-integration/pull-description.jpg b/public/cli/keboola-as-code/github-integration/pull-description.jpg new file mode 100644 index 000000000..228e28f6d Binary files /dev/null and b/public/cli/keboola-as-code/github-integration/pull-description.jpg differ diff --git a/public/cli/keboola-as-code/structure/directory-example.jpg b/public/cli/keboola-as-code/structure/directory-example.jpg new file mode 100644 index 000000000..d692763b5 Binary files /dev/null and b/public/cli/keboola-as-code/structure/directory-example.jpg differ diff --git a/public/cli/keboola-as-code/structure/directory-orchestration-example.png b/public/cli/keboola-as-code/structure/directory-orchestration-example.png new file mode 100644 index 000000000..c0af546c5 Binary files /dev/null and b/public/cli/keboola-as-code/structure/directory-orchestration-example.png differ diff --git a/public/cli/keboola-as-code/structure/directory-rows-example.jpg b/public/cli/keboola-as-code/structure/directory-rows-example.jpg new file mode 100644 index 000000000..be2dd6393 Binary files /dev/null and b/public/cli/keboola-as-code/structure/directory-rows-example.jpg differ diff --git a/public/cli/keboola-as-code/structure/directory-transformation-example.jpg b/public/cli/keboola-as-code/structure/directory-transformation-example.jpg new file mode 100644 index 000000000..bf8e7ef45 Binary files /dev/null and b/public/cli/keboola-as-code/structure/directory-transformation-example.jpg differ diff --git a/public/cli/keboola-as-code/structure/scheduler-directory.jpg b/public/cli/keboola-as-code/structure/scheduler-directory.jpg new file mode 100644 index 000000000..afe1ae2b4 Binary files /dev/null and b/public/cli/keboola-as-code/structure/scheduler-directory.jpg differ diff --git a/public/cli/keboola-as-code/structure/shared-code-code.jpg b/public/cli/keboola-as-code/structure/shared-code-code.jpg new file mode 100644 index 000000000..633abe80d Binary files /dev/null and b/public/cli/keboola-as-code/structure/shared-code-code.jpg differ diff --git a/public/cli/keboola-as-code/structure/shared-code-directory.jpg b/public/cli/keboola-as-code/structure/shared-code-directory.jpg new file mode 100644 index 000000000..6df81929c Binary files /dev/null and b/public/cli/keboola-as-code/structure/shared-code-directory.jpg differ diff --git a/public/cli/keboola-as-code/structure/shared-code-ui.jpg b/public/cli/keboola-as-code/structure/shared-code-ui.jpg new file mode 100644 index 000000000..0b9bc7fcc Binary files /dev/null and b/public/cli/keboola-as-code/structure/shared-code-ui.jpg differ diff --git a/public/cli/keboola-as-code/structure/variables-directory.jpg b/public/cli/keboola-as-code/structure/variables-directory.jpg new file mode 100644 index 000000000..0b9132957 Binary files /dev/null and b/public/cli/keboola-as-code/structure/variables-directory.jpg differ diff --git a/public/cli/keboola-as-code/structure/variables-ui.jpg b/public/cli/keboola-as-code/structure/variables-ui.jpg new file mode 100644 index 000000000..bf506e17b Binary files /dev/null and b/public/cli/keboola-as-code/structure/variables-ui.jpg differ diff --git a/public/extend/common-interface/manifest-files/column-data-type-override.png b/public/extend/common-interface/manifest-files/column-data-type-override.png new file mode 100644 index 000000000..b38959bd7 Binary files /dev/null and b/public/extend/common-interface/manifest-files/column-data-type-override.png differ diff --git a/public/extend/common-interface/manifest-files/column-data-type-use.png b/public/extend/common-interface/manifest-files/column-data-type-use.png new file mode 100644 index 000000000..b770829e8 Binary files /dev/null and b/public/extend/common-interface/manifest-files/column-data-type-use.png differ diff --git a/public/extend/common-interface/manifest-files/column-data-type.png b/public/extend/common-interface/manifest-files/column-data-type.png new file mode 100644 index 000000000..c82ee625f Binary files /dev/null and b/public/extend/common-interface/manifest-files/column-data-type.png differ diff --git a/public/extend/common-interface/sandbox-output.png b/public/extend/common-interface/sandbox-output.png new file mode 100644 index 000000000..babc31ec1 Binary files /dev/null and b/public/extend/common-interface/sandbox-output.png differ diff --git a/public/extend/component/code-patterns/interface-1-add-component.png b/public/extend/component/code-patterns/interface-1-add-component.png new file mode 100644 index 000000000..878105b85 Binary files /dev/null and b/public/extend/component/code-patterns/interface-1-add-component.png differ diff --git a/public/extend/component/code-patterns/interface-2-schema.png b/public/extend/component/code-patterns/interface-2-schema.png new file mode 100644 index 000000000..62b8207dc Binary files /dev/null and b/public/extend/component/code-patterns/interface-2-schema.png differ diff --git a/public/extend/component/code-patterns/interface-3-supported-list.png b/public/extend/component/code-patterns/interface-3-supported-list.png new file mode 100644 index 000000000..13a19b3e0 Binary files /dev/null and b/public/extend/component/code-patterns/interface-3-supported-list.png differ diff --git a/public/extend/component/code-patterns/interface-4-new-transformation.png b/public/extend/component/code-patterns/interface-4-new-transformation.png new file mode 100644 index 000000000..2b0c149fb Binary files /dev/null and b/public/extend/component/code-patterns/interface-4-new-transformation.png differ diff --git a/public/extend/component/code-patterns/interface-5-edit-component.png b/public/extend/component/code-patterns/interface-5-edit-component.png new file mode 100644 index 000000000..5ba5638be Binary files /dev/null and b/public/extend/component/code-patterns/interface-5-edit-component.png differ diff --git a/public/extend/component/code-patterns/tutorial-1-add-component.png b/public/extend/component/code-patterns/tutorial-1-add-component.png new file mode 100644 index 000000000..878105b85 Binary files /dev/null and b/public/extend/component/code-patterns/tutorial-1-add-component.png differ diff --git a/public/extend/component/code-patterns/tutorial-2-project.png b/public/extend/component/code-patterns/tutorial-2-project.png new file mode 100644 index 000000000..a31c3d1e3 Binary files /dev/null and b/public/extend/component/code-patterns/tutorial-2-project.png differ diff --git a/public/extend/component/code-patterns/tutorial-3-modal.png b/public/extend/component/code-patterns/tutorial-3-modal.png new file mode 100644 index 000000000..f4c1d3646 Binary files /dev/null and b/public/extend/component/code-patterns/tutorial-3-modal.png differ diff --git a/public/extend/component/code-patterns/tutorial-4-new-transformation.png b/public/extend/component/code-patterns/tutorial-4-new-transformation.png new file mode 100644 index 000000000..b9bca4779 Binary files /dev/null and b/public/extend/component/code-patterns/tutorial-4-new-transformation.png differ diff --git a/public/extend/component/deployment/bitbucket-1.png b/public/extend/component/deployment/bitbucket-1.png new file mode 100644 index 000000000..4f98ff54f Binary files /dev/null and b/public/extend/component/deployment/bitbucket-1.png differ diff --git a/public/extend/component/deployment/bitbucket-2.png b/public/extend/component/deployment/bitbucket-2.png new file mode 100644 index 000000000..58b8eacd3 Binary files /dev/null and b/public/extend/component/deployment/bitbucket-2.png differ diff --git a/public/extend/component/deployment/bitbucket-3.png b/public/extend/component/deployment/bitbucket-3.png new file mode 100644 index 000000000..5204470d5 Binary files /dev/null and b/public/extend/component/deployment/bitbucket-3.png differ diff --git a/public/extend/component/deployment/configuration-sample.png b/public/extend/component/deployment/configuration-sample.png new file mode 100644 index 000000000..1d02e66ea Binary files /dev/null and b/public/extend/component/deployment/configuration-sample.png differ diff --git a/public/extend/component/deployment/deploy-config-1.png b/public/extend/component/deployment/deploy-config-1.png new file mode 100644 index 000000000..be05fd9a7 Binary files /dev/null and b/public/extend/component/deployment/deploy-config-1.png differ diff --git a/public/extend/component/deployment/deploy-config-2.png b/public/extend/component/deployment/deploy-config-2.png new file mode 100644 index 000000000..1812233ed Binary files /dev/null and b/public/extend/component/deployment/deploy-config-2.png differ diff --git a/public/extend/component/deployment/deploy-config-3.png b/public/extend/component/deployment/deploy-config-3.png new file mode 100644 index 000000000..e4066e6a3 Binary files /dev/null and b/public/extend/component/deployment/deploy-config-3.png differ diff --git a/public/extend/component/deployment/deploy-final.png b/public/extend/component/deployment/deploy-final.png new file mode 100644 index 000000000..dc0c78133 Binary files /dev/null and b/public/extend/component/deployment/deploy-final.png differ diff --git a/public/extend/component/deployment/deploy-log-1.png b/public/extend/component/deployment/deploy-log-1.png new file mode 100644 index 000000000..464d98600 Binary files /dev/null and b/public/extend/component/deployment/deploy-log-1.png differ diff --git a/public/extend/component/deployment/deploy-log-2.png b/public/extend/component/deployment/deploy-log-2.png new file mode 100644 index 000000000..b6ff72791 Binary files /dev/null and b/public/extend/component/deployment/deploy-log-2.png differ diff --git a/public/extend/component/deployment/gitlab-1.png b/public/extend/component/deployment/gitlab-1.png new file mode 100644 index 000000000..a57e64e29 Binary files /dev/null and b/public/extend/component/deployment/gitlab-1.png differ diff --git a/public/extend/component/deployment/gitlab-2.png b/public/extend/component/deployment/gitlab-2.png new file mode 100644 index 000000000..9e17fa6ca Binary files /dev/null and b/public/extend/component/deployment/gitlab-2.png differ diff --git a/public/extend/component/dynamic-mapping.png b/public/extend/component/dynamic-mapping.png new file mode 100644 index 000000000..189a865d6 Binary files /dev/null and b/public/extend/component/dynamic-mapping.png differ diff --git a/public/extend/component/running/input-configuration.png b/public/extend/component/running/input-configuration.png new file mode 100644 index 000000000..9d5f392e1 Binary files /dev/null and b/public/extend/component/running/input-configuration.png differ diff --git a/public/extend/component/running/sandbox-data.png b/public/extend/component/running/sandbox-data.png new file mode 100644 index 000000000..417a8e6f2 Binary files /dev/null and b/public/extend/component/running/sandbox-data.png differ diff --git a/public/extend/component/running/sandbox-progress.png b/public/extend/component/running/sandbox-progress.png new file mode 100644 index 000000000..58867dd5d Binary files /dev/null and b/public/extend/component/running/sandbox-progress.png differ diff --git a/public/extend/component/tutorial/component-configuration.png b/public/extend/component/tutorial/component-configuration.png new file mode 100644 index 000000000..dac4cbb6f Binary files /dev/null and b/public/extend/component/tutorial/component-configuration.png differ diff --git a/public/extend/component/tutorial/component-deployed.png b/public/extend/component/tutorial/component-deployed.png new file mode 100644 index 000000000..eb01d080c Binary files /dev/null and b/public/extend/component/tutorial/component-deployed.png differ diff --git a/public/extend/component/tutorial/component-generator.png b/public/extend/component/tutorial/component-generator.png new file mode 100644 index 000000000..23afcf891 Binary files /dev/null and b/public/extend/component/tutorial/component-generator.png differ diff --git a/public/extend/component/tutorial/configuration-1.png b/public/extend/component/tutorial/configuration-1.png new file mode 100644 index 000000000..9524e3369 Binary files /dev/null and b/public/extend/component/tutorial/configuration-1.png differ diff --git a/public/extend/component/tutorial/configuration-2.png b/public/extend/component/tutorial/configuration-2.png new file mode 100644 index 000000000..500901afa Binary files /dev/null and b/public/extend/component/tutorial/configuration-2.png differ diff --git a/public/extend/component/tutorial/configuration-3.png b/public/extend/component/tutorial/configuration-3.png new file mode 100644 index 000000000..bbe684a3f Binary files /dev/null and b/public/extend/component/tutorial/configuration-3.png differ diff --git a/public/extend/component/tutorial/configuration-4.png b/public/extend/component/tutorial/configuration-4.png new file mode 100644 index 000000000..ec9a05065 Binary files /dev/null and b/public/extend/component/tutorial/configuration-4.png differ diff --git a/public/extend/component/tutorial/configuration-sample.png b/public/extend/component/tutorial/configuration-sample.png new file mode 100644 index 000000000..fd9eb1907 Binary files /dev/null and b/public/extend/component/tutorial/configuration-sample.png differ diff --git a/public/extend/component/tutorial/create-component-1.png b/public/extend/component/tutorial/create-component-1.png new file mode 100644 index 000000000..5ab40e12b Binary files /dev/null and b/public/extend/component/tutorial/create-component-1.png differ diff --git a/public/extend/component/tutorial/create-component-2.png b/public/extend/component/tutorial/create-component-2.png new file mode 100644 index 000000000..3301f7e05 Binary files /dev/null and b/public/extend/component/tutorial/create-component-2.png differ diff --git a/public/extend/component/tutorial/debug-1.png b/public/extend/component/tutorial/debug-1.png new file mode 100644 index 000000000..dadeb9caa Binary files /dev/null and b/public/extend/component/tutorial/debug-1.png differ diff --git a/public/extend/component/tutorial/debug-2.png b/public/extend/component/tutorial/debug-2.png new file mode 100644 index 000000000..30f58d6c9 Binary files /dev/null and b/public/extend/component/tutorial/debug-2.png differ diff --git a/public/extend/component/tutorial/debug-3.png b/public/extend/component/tutorial/debug-3.png new file mode 100644 index 000000000..2c5896832 Binary files /dev/null and b/public/extend/component/tutorial/debug-3.png differ diff --git a/public/extend/component/tutorial/debug-4.png b/public/extend/component/tutorial/debug-4.png new file mode 100644 index 000000000..9b67d51ba Binary files /dev/null and b/public/extend/component/tutorial/debug-4.png differ diff --git a/public/extend/component/tutorial/gh-build-1.png b/public/extend/component/tutorial/gh-build-1.png new file mode 100644 index 000000000..c090286c2 Binary files /dev/null and b/public/extend/component/tutorial/gh-build-1.png differ diff --git a/public/extend/component/tutorial/gh-build-2.png b/public/extend/component/tutorial/gh-build-2.png new file mode 100644 index 000000000..fceafd004 Binary files /dev/null and b/public/extend/component/tutorial/gh-build-2.png differ diff --git a/public/extend/component/tutorial/github-repository.png b/public/extend/component/tutorial/github-repository.png new file mode 100644 index 000000000..92cadbf69 Binary files /dev/null and b/public/extend/component/tutorial/github-repository.png differ diff --git a/public/extend/component/tutorial/hello-world.png b/public/extend/component/tutorial/hello-world.png new file mode 100644 index 000000000..fe2568dcb Binary files /dev/null and b/public/extend/component/tutorial/hello-world.png differ diff --git a/public/extend/component/tutorial/input-mapping-1.png b/public/extend/component/tutorial/input-mapping-1.png new file mode 100644 index 000000000..c06fb9a02 Binary files /dev/null and b/public/extend/component/tutorial/input-mapping-1.png differ diff --git a/public/extend/component/tutorial/input-mapping-2.png b/public/extend/component/tutorial/input-mapping-2.png new file mode 100644 index 000000000..cffc19777 Binary files /dev/null and b/public/extend/component/tutorial/input-mapping-2.png differ diff --git a/public/extend/component/tutorial/input-mapping-3.png b/public/extend/component/tutorial/input-mapping-3.png new file mode 100644 index 000000000..8b45d1179 Binary files /dev/null and b/public/extend/component/tutorial/input-mapping-3.png differ diff --git a/public/extend/component/tutorial/input-mapping-4.png b/public/extend/component/tutorial/input-mapping-4.png new file mode 100644 index 000000000..b034be600 Binary files /dev/null and b/public/extend/component/tutorial/input-mapping-4.png differ diff --git a/public/extend/component/tutorial/join-vendor.png b/public/extend/component/tutorial/join-vendor.png new file mode 100644 index 000000000..ac5bc139f Binary files /dev/null and b/public/extend/component/tutorial/join-vendor.png differ diff --git a/public/extend/component/tutorial/output-mapping-1.png b/public/extend/component/tutorial/output-mapping-1.png new file mode 100644 index 000000000..d8ae0bd03 Binary files /dev/null and b/public/extend/component/tutorial/output-mapping-1.png differ diff --git a/public/extend/component/tutorial/output-mapping-2.png b/public/extend/component/tutorial/output-mapping-2.png new file mode 100644 index 000000000..ad9c1089f Binary files /dev/null and b/public/extend/component/tutorial/output-mapping-2.png differ diff --git a/public/extend/component/tutorial/processors-1.png b/public/extend/component/tutorial/processors-1.png new file mode 100644 index 000000000..3af231056 Binary files /dev/null and b/public/extend/component/tutorial/processors-1.png differ diff --git a/public/extend/component/tutorial/service-account-1.png b/public/extend/component/tutorial/service-account-1.png new file mode 100644 index 000000000..9bc4dfddf Binary files /dev/null and b/public/extend/component/tutorial/service-account-1.png differ diff --git a/public/extend/component/tutorial/service-account-2.png b/public/extend/component/tutorial/service-account-2.png new file mode 100644 index 000000000..fb5e0cf99 Binary files /dev/null and b/public/extend/component/tutorial/service-account-2.png differ diff --git a/public/extend/component/tutorial/service-account-3.png b/public/extend/component/tutorial/service-account-3.png new file mode 100644 index 000000000..ce143da53 Binary files /dev/null and b/public/extend/component/tutorial/service-account-3.png differ diff --git a/public/extend/component/tutorial/travis-build-1.png b/public/extend/component/tutorial/travis-build-1.png new file mode 100644 index 000000000..24cb081a3 Binary files /dev/null and b/public/extend/component/tutorial/travis-build-1.png differ diff --git a/public/extend/component/tutorial/travis-build-2.png b/public/extend/component/tutorial/travis-build-2.png new file mode 100644 index 000000000..92d95d270 Binary files /dev/null and b/public/extend/component/tutorial/travis-build-2.png differ diff --git a/public/extend/component/ui-options/auth-0.png b/public/extend/component/ui-options/auth-0.png new file mode 100644 index 000000000..5fccc1bca Binary files /dev/null and b/public/extend/component/ui-options/auth-0.png differ diff --git a/public/extend/component/ui-options/auth-1.png b/public/extend/component/ui-options/auth-1.png new file mode 100644 index 000000000..a0fd4be88 Binary files /dev/null and b/public/extend/component/ui-options/auth-1.png differ diff --git a/public/extend/component/ui-options/configuration-schema-1.png b/public/extend/component/ui-options/configuration-schema-1.png new file mode 100644 index 000000000..0b7272323 Binary files /dev/null and b/public/extend/component/ui-options/configuration-schema-1.png differ diff --git a/public/extend/component/ui-options/configuration-schema-2.png b/public/extend/component/ui-options/configuration-schema-2.png new file mode 100644 index 000000000..e53c81a37 Binary files /dev/null and b/public/extend/component/ui-options/configuration-schema-2.png differ diff --git a/public/extend/component/ui-options/configuration.png b/public/extend/component/ui-options/configuration.png new file mode 100644 index 000000000..f4f60d848 Binary files /dev/null and b/public/extend/component/ui-options/configuration.png differ diff --git a/public/extend/component/ui-options/default-configuration/developer-portal-01.png b/public/extend/component/ui-options/default-configuration/developer-portal-01.png new file mode 100644 index 000000000..0145e2c7a Binary files /dev/null and b/public/extend/component/ui-options/default-configuration/developer-portal-01.png differ diff --git a/public/extend/component/ui-options/file-input-0.png b/public/extend/component/ui-options/file-input-0.png new file mode 100644 index 000000000..0d8cd5589 Binary files /dev/null and b/public/extend/component/ui-options/file-input-0.png differ diff --git a/public/extend/component/ui-options/file-input-1.png b/public/extend/component/ui-options/file-input-1.png new file mode 100644 index 000000000..a7dffe99d Binary files /dev/null and b/public/extend/component/ui-options/file-input-1.png differ diff --git a/public/extend/component/ui-options/file-input-2.png b/public/extend/component/ui-options/file-input-2.png new file mode 100644 index 000000000..602a0df30 Binary files /dev/null and b/public/extend/component/ui-options/file-input-2.png differ diff --git a/public/extend/component/ui-options/file-output-0.png b/public/extend/component/ui-options/file-output-0.png new file mode 100644 index 000000000..599c95b96 Binary files /dev/null and b/public/extend/component/ui-options/file-output-0.png differ diff --git a/public/extend/component/ui-options/file-output-1.png b/public/extend/component/ui-options/file-output-1.png new file mode 100644 index 000000000..aaf8ff94f Binary files /dev/null and b/public/extend/component/ui-options/file-output-1.png differ diff --git a/public/extend/component/ui-options/file-output-2.png b/public/extend/component/ui-options/file-output-2.png new file mode 100644 index 000000000..b4ae88ba1 Binary files /dev/null and b/public/extend/component/ui-options/file-output-2.png differ diff --git a/public/extend/component/ui-options/form.png b/public/extend/component/ui-options/form.png new file mode 100644 index 000000000..ad77b25b7 Binary files /dev/null and b/public/extend/component/ui-options/form.png differ diff --git a/public/extend/component/ui-options/processors.png b/public/extend/component/ui-options/processors.png new file mode 100644 index 000000000..ac598b859 Binary files /dev/null and b/public/extend/component/ui-options/processors.png differ diff --git a/public/extend/component/ui-options/table-input-0.png b/public/extend/component/ui-options/table-input-0.png new file mode 100644 index 000000000..e98889781 Binary files /dev/null and b/public/extend/component/ui-options/table-input-0.png differ diff --git a/public/extend/component/ui-options/table-input-1.png b/public/extend/component/ui-options/table-input-1.png new file mode 100644 index 000000000..03e4ffe3a Binary files /dev/null and b/public/extend/component/ui-options/table-input-1.png differ diff --git a/public/extend/component/ui-options/table-input-2.png b/public/extend/component/ui-options/table-input-2.png new file mode 100644 index 000000000..28c1d73a9 Binary files /dev/null and b/public/extend/component/ui-options/table-input-2.png differ diff --git a/public/extend/component/ui-options/table-output-0.png b/public/extend/component/ui-options/table-output-0.png new file mode 100644 index 000000000..188d2c3d8 Binary files /dev/null and b/public/extend/component/ui-options/table-output-0.png differ diff --git a/public/extend/component/ui-options/table-output-1.png b/public/extend/component/ui-options/table-output-1.png new file mode 100644 index 000000000..48e0e4aa1 Binary files /dev/null and b/public/extend/component/ui-options/table-output-1.png differ diff --git a/public/extend/component/ui-options/table-output-2.png b/public/extend/component/ui-options/table-output-2.png new file mode 100644 index 000000000..343c5c00c Binary files /dev/null and b/public/extend/component/ui-options/table-output-2.png differ diff --git a/public/extend/component/ui-options/ui-examples/checkbox.png b/public/extend/component/ui-options/ui-examples/checkbox.png new file mode 100644 index 000000000..cc5b5f0cb Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/checkbox.png differ diff --git a/public/extend/component/ui-options/ui-examples/code_editor.png b/public/extend/component/ui-options/ui-examples/code_editor.png new file mode 100644 index 000000000..c48ad8c74 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/code_editor.png differ diff --git a/public/extend/component/ui-options/ui-examples/creatable_select.gif b/public/extend/component/ui-options/ui-examples/creatable_select.gif new file mode 100644 index 000000000..18860877f Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/creatable_select.gif differ diff --git a/public/extend/component/ui-options/ui-examples/det_period.png b/public/extend/component/ui-options/ui-examples/det_period.png new file mode 100644 index 000000000..9d9c52b50 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/det_period.png differ diff --git a/public/extend/component/ui-options/ui-examples/dynamic_dropdown_multi.gif b/public/extend/component/ui-options/ui-examples/dynamic_dropdown_multi.gif new file mode 100644 index 000000000..699725d5a Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/dynamic_dropdown_multi.gif differ diff --git a/public/extend/component/ui-options/ui-examples/dynamic_sel.gif b/public/extend/component/ui-options/ui-examples/dynamic_sel.gif new file mode 100644 index 000000000..5e207a05d Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/dynamic_sel.gif differ diff --git a/public/extend/component/ui-options/ui-examples/generic-button.gif b/public/extend/component/ui-options/ui-examples/generic-button.gif new file mode 100644 index 000000000..b1c4fe45a Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/generic-button.gif differ diff --git a/public/extend/component/ui-options/ui-examples/load_type.png b/public/extend/component/ui-options/ui-examples/load_type.png new file mode 100644 index 000000000..b462eecf6 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/load_type.png differ diff --git a/public/extend/component/ui-options/ui-examples/loading_options_block.png b/public/extend/component/ui-options/ui-examples/loading_options_block.png new file mode 100644 index 000000000..3ef882f48 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/loading_options_block.png differ diff --git a/public/extend/component/ui-options/ui-examples/multi_select.png b/public/extend/component/ui-options/ui-examples/multi_select.png new file mode 100644 index 000000000..934f920d5 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/multi_select.png differ diff --git a/public/extend/component/ui-options/ui-examples/optional_block_array.gif b/public/extend/component/ui-options/ui-examples/optional_block_array.gif new file mode 100644 index 000000000..e78adaed1 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/optional_block_array.gif differ diff --git a/public/extend/component/ui-options/ui-examples/password.png b/public/extend/component/ui-options/ui-examples/password.png new file mode 100644 index 000000000..da72d705b Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/password.png differ diff --git a/public/extend/component/ui-options/ui-examples/single-drop.gif b/public/extend/component/ui-options/ui-examples/single-drop.gif new file mode 100644 index 000000000..f2a331615 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/single-drop.gif differ diff --git a/public/extend/component/ui-options/ui-examples/status_icons.png b/public/extend/component/ui-options/ui-examples/status_icons.png new file mode 100644 index 000000000..f725c4872 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/status_icons.png differ diff --git a/public/extend/component/ui-options/ui-examples/test_connection.png b/public/extend/component/ui-options/ui-examples/test_connection.png new file mode 100644 index 000000000..a1020d381 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/test_connection.png differ diff --git a/public/extend/component/ui-options/ui-examples/tooltip_normal.png b/public/extend/component/ui-options/ui-examples/tooltip_normal.png new file mode 100644 index 000000000..3b1295200 Binary files /dev/null and b/public/extend/component/ui-options/ui-examples/tooltip_normal.png differ diff --git a/public/extend/data.zip b/public/extend/data.zip new file mode 100644 index 000000000..0dffb3455 Binary files /dev/null and b/public/extend/data.zip differ diff --git a/public/extend/generic-extractor/configuration.png b/public/extend/generic-extractor/configuration.png new file mode 100644 index 000000000..a2b613521 Binary files /dev/null and b/public/extend/generic-extractor/configuration.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/api_key.png b/public/extend/generic-extractor/configuration/api/authentication/api_key.png new file mode 100644 index 000000000..e3b86952d Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/api_key.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/auth_ui.png b/public/extend/generic-extractor/configuration/api/authentication/auth_ui.png new file mode 100644 index 000000000..0a85af673 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/auth_ui.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/basic.png b/public/extend/generic-extractor/configuration/api/authentication/basic.png new file mode 100644 index 000000000..92af3c7ab Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/basic.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/bearer.png b/public/extend/generic-extractor/configuration/api/authentication/bearer.png new file mode 100644 index 000000000..c686dd510 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/bearer.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/login.png b/public/extend/generic-extractor/configuration/api/authentication/login.png new file mode 100644 index 000000000..49d15e838 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/login.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/oauth10-diagram.png b/public/extend/generic-extractor/configuration/api/authentication/oauth10-diagram.png new file mode 100644 index 000000000..8ed840a11 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/oauth10-diagram.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/oauth20-diagram.png b/public/extend/generic-extractor/configuration/api/authentication/oauth20-diagram.png new file mode 100644 index 000000000..bee415344 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/oauth20-diagram.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-console.png b/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-console.png new file mode 100644 index 000000000..83c16a728 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-console.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-1.png b/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-1.png new file mode 100644 index 000000000..88117a4ed Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-1.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-2.png b/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-2.png new file mode 100644 index 000000000..742cfb528 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-2.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-3.png b/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-3.png new file mode 100644 index 000000000..5e56f4399 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/oauth20-login-playground-3.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/oauth_cc.png b/public/extend/generic-extractor/configuration/api/authentication/oauth_cc.png new file mode 100644 index 000000000..eb301e270 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/oauth_cc.png differ diff --git a/public/extend/generic-extractor/configuration/api/authentication/query.png b/public/extend/generic-extractor/configuration/api/authentication/query.png new file mode 100644 index 000000000..6bba5c4d6 Binary files /dev/null and b/public/extend/generic-extractor/configuration/api/authentication/query.png differ diff --git a/public/extend/generic-extractor/configuration/ui_switch.png b/public/extend/generic-extractor/configuration/ui_switch.png new file mode 100644 index 000000000..83152c49a Binary files /dev/null and b/public/extend/generic-extractor/configuration/ui_switch.png differ diff --git a/public/extend/generic-extractor/events.png b/public/extend/generic-extractor/events.png new file mode 100644 index 000000000..3609158ca Binary files /dev/null and b/public/extend/generic-extractor/events.png differ diff --git a/public/extend/generic-extractor/function_eval.gif b/public/extend/generic-extractor/function_eval.gif new file mode 100644 index 000000000..4513c09cc Binary files /dev/null and b/public/extend/generic-extractor/function_eval.gif differ diff --git a/public/extend/generic-extractor/functions.png b/public/extend/generic-extractor/functions.png new file mode 100644 index 000000000..0a893a9f2 Binary files /dev/null and b/public/extend/generic-extractor/functions.png differ diff --git a/public/extend/generic-extractor/generic-intro.png b/public/extend/generic-extractor/generic-intro.png new file mode 100644 index 000000000..c50c3e4b5 Binary files /dev/null and b/public/extend/generic-extractor/generic-intro.png differ diff --git a/public/extend/generic-extractor/schema-test.png b/public/extend/generic-extractor/schema-test.png new file mode 100644 index 000000000..f141f29f7 Binary files /dev/null and b/public/extend/generic-extractor/schema-test.png differ diff --git a/public/extend/generic-extractor/template-1.png b/public/extend/generic-extractor/template-1.png new file mode 100644 index 000000000..bce5ad800 Binary files /dev/null and b/public/extend/generic-extractor/template-1.png differ diff --git a/public/extend/generic-extractor/tutorial/2_child.png b/public/extend/generic-extractor/tutorial/2_child.png new file mode 100644 index 000000000..2678c3f85 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/2_child.png differ diff --git a/public/extend/generic-extractor/tutorial/base_configuration.png b/public/extend/generic-extractor/tutorial/base_configuration.png new file mode 100644 index 000000000..2f5ca6297 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/base_configuration.png differ diff --git a/public/extend/generic-extractor/tutorial/child_debug.png b/public/extend/generic-extractor/tutorial/child_debug.png new file mode 100644 index 000000000..926488ff5 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/child_debug.png differ diff --git a/public/extend/generic-extractor/tutorial/child_endpoint.png b/public/extend/generic-extractor/tutorial/child_endpoint.png new file mode 100644 index 000000000..b431a9b81 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/child_endpoint.png differ diff --git a/public/extend/generic-extractor/tutorial/config-1.png b/public/extend/generic-extractor/tutorial/config-1.png new file mode 100644 index 000000000..55f407d96 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/config-1.png differ diff --git a/public/extend/generic-extractor/tutorial/configuration-schema.svg b/public/extend/generic-extractor/tutorial/configuration-schema.svg new file mode 100644 index 000000000..0c010c242 --- /dev/null +++ b/public/extend/generic-extractor/tutorial/configuration-schema.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/public/extend/generic-extractor/tutorial/create_endpoint_child.png b/public/extend/generic-extractor/tutorial/create_endpoint_child.png new file mode 100644 index 000000000..07dbf2f49 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/create_endpoint_child.png differ diff --git a/public/extend/generic-extractor/tutorial/create_mapping.png b/public/extend/generic-extractor/tutorial/create_mapping.png new file mode 100644 index 000000000..8fa62ba38 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/create_mapping.png differ diff --git a/public/extend/generic-extractor/tutorial/create_mapping_toggle.png b/public/extend/generic-extractor/tutorial/create_mapping_toggle.png new file mode 100644 index 000000000..bf1c4d65f Binary files /dev/null and b/public/extend/generic-extractor/tutorial/create_mapping_toggle.png differ diff --git a/public/extend/generic-extractor/tutorial/data_selector.png b/public/extend/generic-extractor/tutorial/data_selector.png new file mode 100644 index 000000000..91083e09b Binary files /dev/null and b/public/extend/generic-extractor/tutorial/data_selector.png differ diff --git a/public/extend/generic-extractor/tutorial/img.png b/public/extend/generic-extractor/tutorial/img.png new file mode 100644 index 000000000..02755a43f Binary files /dev/null and b/public/extend/generic-extractor/tutorial/img.png differ diff --git a/public/extend/generic-extractor/tutorial/img_1.png b/public/extend/generic-extractor/tutorial/img_1.png new file mode 100644 index 000000000..70e32bd77 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/img_1.png differ diff --git a/public/extend/generic-extractor/tutorial/job-1.png b/public/extend/generic-extractor/tutorial/job-1.png new file mode 100644 index 000000000..4b784849f Binary files /dev/null and b/public/extend/generic-extractor/tutorial/job-1.png differ diff --git a/public/extend/generic-extractor/tutorial/job-2.png b/public/extend/generic-extractor/tutorial/job-2.png new file mode 100644 index 000000000..1b113919a Binary files /dev/null and b/public/extend/generic-extractor/tutorial/job-2.png differ diff --git a/public/extend/generic-extractor/tutorial/job-table-1.png b/public/extend/generic-extractor/tutorial/job-table-1.png new file mode 100644 index 000000000..fedfc0f45 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/job-table-1.png differ diff --git a/public/extend/generic-extractor/tutorial/job-table-2.png b/public/extend/generic-extractor/tutorial/job-table-2.png new file mode 100644 index 000000000..49237514c Binary files /dev/null and b/public/extend/generic-extractor/tutorial/job-table-2.png differ diff --git a/public/extend/generic-extractor/tutorial/mailchimp-api-docs-1.png b/public/extend/generic-extractor/tutorial/mailchimp-api-docs-1.png new file mode 100644 index 000000000..8e61aaf81 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/mailchimp-api-docs-1.png differ diff --git a/public/extend/generic-extractor/tutorial/mailchimp-api-docs-2.png b/public/extend/generic-extractor/tutorial/mailchimp-api-docs-2.png new file mode 100644 index 000000000..1ae66b621 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/mailchimp-api-docs-2.png differ diff --git a/public/extend/generic-extractor/tutorial/mapping_all.png b/public/extend/generic-extractor/tutorial/mapping_all.png new file mode 100644 index 000000000..f7b207ed3 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/mapping_all.png differ diff --git a/public/extend/generic-extractor/tutorial/new_endpoint.png b/public/extend/generic-extractor/tutorial/new_endpoint.png new file mode 100644 index 000000000..aeec1bfba Binary files /dev/null and b/public/extend/generic-extractor/tutorial/new_endpoint.png differ diff --git a/public/extend/generic-extractor/tutorial/new_endpoint_modal.png b/public/extend/generic-extractor/tutorial/new_endpoint_modal.png new file mode 100644 index 000000000..6425aaa54 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/new_endpoint_modal.png differ diff --git a/public/extend/generic-extractor/tutorial/pagination.png b/public/extend/generic-extractor/tutorial/pagination.png new file mode 100644 index 000000000..3b892a17e Binary files /dev/null and b/public/extend/generic-extractor/tutorial/pagination.png differ diff --git a/public/extend/generic-extractor/tutorial/sub-resources-docs.png b/public/extend/generic-extractor/tutorial/sub-resources-docs.png new file mode 100644 index 000000000..2ec676187 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/sub-resources-docs.png differ diff --git a/public/extend/generic-extractor/tutorial/table-campaigns-sample.png b/public/extend/generic-extractor/tutorial/table-campaigns-sample.png new file mode 100644 index 000000000..fd906395e Binary files /dev/null and b/public/extend/generic-extractor/tutorial/table-campaigns-sample.png differ diff --git a/public/extend/generic-extractor/tutorial/test_endpoint.png b/public/extend/generic-extractor/tutorial/test_endpoint.png new file mode 100644 index 000000000..64cd43881 Binary files /dev/null and b/public/extend/generic-extractor/tutorial/test_endpoint.png differ diff --git a/public/extend/generic-extractor/ui.png b/public/extend/generic-extractor/ui.png new file mode 100644 index 000000000..6cb30b30b Binary files /dev/null and b/public/extend/generic-extractor/ui.png differ diff --git a/public/extend/job-queue/docker-runner.svg b/public/extend/job-queue/docker-runner.svg new file mode 100644 index 000000000..2968f7a7e --- /dev/null +++ b/public/extend/job-queue/docker-runner.svg @@ -0,0 +1,2 @@ + +
Job Queue
Job Queue
Isolated Container
Isolated Container
Storage API
Storage API
Project Storage
Project Storage
Job Configuration
[Not supported by viewer]
Storage Input
Storage Input
Storage Output
Storage Output
State
State
Parameters
Parameters
Input Tables
Input Tables
/data/in/tables/
/data/in/files/
[Not supported by viewer]
Output Tables
Output Tables
Output Files
Output Files
/data/out/tables/
/data/out/files/
[Not supported by viewer]
/data/config.json
/data/in/state.json
[Not supported by viewer]
Component Definition
Component Definition
Component Configuration
Component Configuration
Pull Docker Image
Pull Docker Image
stdout / stderr
stdout / stderr
Run Container
Run Container
/data/out/state.json
<div>/data/out/state.json</div>
Events
Events
Input Files
Input Files


<div><br></div><div><br></div>
Job Result
[Not supported by viewer]
Status
Status
Exit code
Exit code
\ No newline at end of file diff --git a/public/extend/publish/approve.png b/public/extend/publish/approve.png new file mode 100644 index 000000000..3b9e2bc77 Binary files /dev/null and b/public/extend/publish/approve.png differ diff --git a/public/extend/publish/schema-bad.png b/public/extend/publish/schema-bad.png new file mode 100644 index 000000000..ed2450754 Binary files /dev/null and b/public/extend/publish/schema-bad.png differ diff --git a/public/extend/publish/schema-good.png b/public/extend/publish/schema-good.png new file mode 100644 index 000000000..8082071b7 Binary files /dev/null and b/public/extend/publish/schema-good.png differ diff --git a/public/integrate/artifacts/artifacts-tutorial-1.png b/public/integrate/artifacts/artifacts-tutorial-1.png new file mode 100644 index 000000000..b35de9b1a Binary files /dev/null and b/public/integrate/artifacts/artifacts-tutorial-1.png differ diff --git a/public/integrate/artifacts/artifacts-tutorial-2.png b/public/integrate/artifacts/artifacts-tutorial-2.png new file mode 100644 index 000000000..3b78098be Binary files /dev/null and b/public/integrate/artifacts/artifacts-tutorial-2.png differ diff --git a/public/integrate/artifacts/artifacts-tutorial-3.png b/public/integrate/artifacts/artifacts-tutorial-3.png new file mode 100644 index 000000000..d9ba47847 Binary files /dev/null and b/public/integrate/artifacts/artifacts-tutorial-3.png differ diff --git a/public/integrate/artifacts/artifacts-tutorial-4.png b/public/integrate/artifacts/artifacts-tutorial-4.png new file mode 100644 index 000000000..ae120cc96 Binary files /dev/null and b/public/integrate/artifacts/artifacts-tutorial-4.png differ diff --git a/public/integrate/data-streams/push_data.drawio.png b/public/integrate/data-streams/push_data.drawio.png new file mode 100644 index 000000000..19abce2f6 Binary files /dev/null and b/public/integrate/data-streams/push_data.drawio.png differ diff --git a/public/integrate/data-streams/tutorial/gh-settings-webhook-add.png b/public/integrate/data-streams/tutorial/gh-settings-webhook-add.png new file mode 100644 index 000000000..d98d32b94 Binary files /dev/null and b/public/integrate/data-streams/tutorial/gh-settings-webhook-add.png differ diff --git a/public/integrate/data-streams/tutorial/gh-settings-webhook-individual-events.png b/public/integrate/data-streams/tutorial/gh-settings-webhook-individual-events.png new file mode 100644 index 000000000..43908f2d6 Binary files /dev/null and b/public/integrate/data-streams/tutorial/gh-settings-webhook-individual-events.png differ diff --git a/public/integrate/data-streams/tutorial/gh-settings-webhook-issues.png b/public/integrate/data-streams/tutorial/gh-settings-webhook-issues.png new file mode 100644 index 000000000..0ab61f8f9 Binary files /dev/null and b/public/integrate/data-streams/tutorial/gh-settings-webhook-issues.png differ diff --git a/public/integrate/data-streams/tutorial/gh-settings-webhook.png b/public/integrate/data-streams/tutorial/gh-settings-webhook.png new file mode 100644 index 000000000..4b9fecec2 Binary files /dev/null and b/public/integrate/data-streams/tutorial/gh-settings-webhook.png differ diff --git a/public/integrate/data-streams/tutorial/gh-tabs.png b/public/integrate/data-streams/tutorial/gh-tabs.png new file mode 100644 index 000000000..a78cb9d47 Binary files /dev/null and b/public/integrate/data-streams/tutorial/gh-tabs.png differ diff --git a/public/integrate/data-streams/tutorial/github_webhook_export_file.png b/public/integrate/data-streams/tutorial/github_webhook_export_file.png new file mode 100644 index 000000000..0440b7d7d Binary files /dev/null and b/public/integrate/data-streams/tutorial/github_webhook_export_file.png differ diff --git a/public/integrate/data-streams/tutorial/github_webhook_export_table.png b/public/integrate/data-streams/tutorial/github_webhook_export_table.png new file mode 100644 index 000000000..c13dac3a4 Binary files /dev/null and b/public/integrate/data-streams/tutorial/github_webhook_export_table.png differ diff --git a/public/integrate/data-streams/tutorial/github_webhook_export_table_data.png b/public/integrate/data-streams/tutorial/github_webhook_export_table_data.png new file mode 100644 index 000000000..44192eb0a Binary files /dev/null and b/public/integrate/data-streams/tutorial/github_webhook_export_table_data.png differ diff --git a/public/integrate/data-streams/tutorial/github_webhook_export_token.png b/public/integrate/data-streams/tutorial/github_webhook_export_token.png new file mode 100644 index 000000000..b5ecdce29 Binary files /dev/null and b/public/integrate/data-streams/tutorial/github_webhook_export_token.png differ diff --git a/public/integrate/data-streams/tutorial/table.png b/public/integrate/data-streams/tutorial/table.png new file mode 100644 index 000000000..7e39f0b9b Binary files /dev/null and b/public/integrate/data-streams/tutorial/table.png differ diff --git a/public/integrate/data-streams/tutorial/token.png b/public/integrate/data-streams/tutorial/token.png new file mode 100644 index 000000000..d6b94c7f2 Binary files /dev/null and b/public/integrate/data-streams/tutorial/token.png differ diff --git a/public/integrate/database/ssh-tunnel.jpg b/public/integrate/database/ssh-tunnel.jpg new file mode 100644 index 000000000..d9d9d38a0 Binary files /dev/null and b/public/integrate/database/ssh-tunnel.jpg differ diff --git a/public/integrate/jobs/states.png b/public/integrate/jobs/states.png new file mode 100644 index 000000000..b3e0e6e95 Binary files /dev/null and b/public/integrate/jobs/states.png differ diff --git a/public/integrate/storage/api/async-import-handling.svg b/public/integrate/storage/api/async-import-handling.svg new file mode 100644 index 000000000..777e93372 --- /dev/null +++ b/public/integrate/storage/api/async-import-handling.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/public/integrate/storage/new-table.csv b/public/integrate/storage/new-table.csv new file mode 100644 index 000000000..8dbb6c464 --- /dev/null +++ b/public/integrate/storage/new-table.csv @@ -0,0 +1,5 @@ +"id","secondCol" +"1","a" +"2","b" +"3","c" +"4","d" \ No newline at end of file diff --git a/public/integrate/storage/sys.c-table-importer.test-config.csv b/public/integrate/storage/sys.c-table-importer.test-config.csv new file mode 100644 index 000000000..c27017d80 --- /dev/null +++ b/public/integrate/storage/sys.c-table-importer.test-config.csv @@ -0,0 +1,2 @@ +"table","primaryKey","incremental","enclosure","delimiter","escapedBy","tag","rowId" +"in.c-main.new-table","","0","""",",","","new-data","1" diff --git a/public/integrate/variables/countries.csv b/public/integrate/variables/countries.csv new file mode 100644 index 000000000..3ff3dec77 --- /dev/null +++ b/public/integrate/variables/countries.csv @@ -0,0 +1,21 @@ +"COUNTRY","CARS" +"Belgium","6293781" +"Finland","3358232" +"Italy","41393877" +"Romania","6541260" +"Turkey","20193915" +"Bulgaria","2823705" +"France","38720798" +"Netherlands","8977994" +"Russia","42201083" +"Ukraine","8655700" +"Czech Republic","5116750" +"Germany","47418800" +"Poland","20671278" +"Spain","27528877" +"United Kingdom","33792233" +"Azerbaijan","1080912" +"Denmark","2723040" +"Hungary","3393075" +"Portugal","5650428" +"Sweden","5126572" diff --git a/public/integrate/variables/tutorial-1.png b/public/integrate/variables/tutorial-1.png new file mode 100644 index 000000000..736929496 Binary files /dev/null and b/public/integrate/variables/tutorial-1.png differ diff --git a/public/integrate/variables/tutorial-2.png b/public/integrate/variables/tutorial-2.png new file mode 100644 index 000000000..d0730e343 Binary files /dev/null and b/public/integrate/variables/tutorial-2.png differ diff --git a/public/integrate/variables/variables.svg b/public/integrate/variables/variables.svg new file mode 100644 index 000000000..5433c2a1c --- /dev/null +++ b/public/integrate/variables/variables.svg @@ -0,0 +1,3 @@ + + +
variables_id
variables_id
variable_values_id
variable_values_id
Main Configuration
(vendor.component)
Main Configuration...
Variables Configuration
(keboola.variables)
Variables Configurat...
config
config
variableValuesId
variableValuesId
Orchestration
Orchestration
Variable Values Row
Variable Values...
config
config
variableValuesId
variableValuesId
Run Orchestration
Run O...
config
config
variableValuesId
variableValuesId
Run Configuration
Run C...
Viewer does not support full SVG 1.1
\ No newline at end of file diff --git a/public/kbc_structure.png b/public/kbc_structure.png new file mode 100644 index 000000000..3f859daba Binary files /dev/null and b/public/kbc_structure.png differ diff --git a/public/overview/api/apiary-console.png b/public/overview/api/apiary-console.png new file mode 100644 index 000000000..1965d06e3 Binary files /dev/null and b/public/overview/api/apiary-console.png differ diff --git a/public/overview/api/postman-import.png b/public/overview/api/postman-import.png new file mode 100644 index 000000000..12333eb04 Binary files /dev/null and b/public/overview/api/postman-import.png differ diff --git a/public/overview/encryption-1.png b/public/overview/encryption-1.png new file mode 100644 index 000000000..9d28c0707 Binary files /dev/null and b/public/overview/encryption-1.png differ diff --git a/public/overview/encryption-2.png b/public/overview/encryption-2.png new file mode 100644 index 000000000..51bb3da41 Binary files /dev/null and b/public/overview/encryption-2.png differ diff --git a/scripts/migrate-devdocs.mjs b/scripts/migrate-devdocs.mjs new file mode 100644 index 000000000..74ba4fd91 --- /dev/null +++ b/scripts/migrate-devdocs.mjs @@ -0,0 +1,494 @@ +#!/usr/bin/env node +/** + * migrate-devdocs.mjs — PHASE 1: full programmatic migration of developers.keboola.com + * (Jekyll) into this Astro/Starlight repo. ONE deterministic run, zero hand edits after. + * + * Per Jordan (2026-07-15): the migration must be a script — verifiable by validating + * this code, not by re-reading 175 pages. Topic re-organization happens later + * (phase 2) per PLACEMENT-MAP.md; here paths are preserved 1:1 (except cli/, see below). + * + * Usage: + * git archive devdocs/main | tar -x -C + * node scripts/migrate-devdocs.mjs [--dry] + * + * What it does, in order: + * 1. PORT every dev page except SKIP (pages already woven into help by open weave + * PRs #1019/#1022/#1023 or dead chrome). Slugs = permalinks, 1:1. + * 2. cli/** is slug-remapped to cli/keboola-as-code/** ("Keboola as Code CLI") and its + * index gets a scripted :::caution[Deprecated] banner → kbagent CLI (Jordan's call). + * 3. Jekyll→Starlight conversion incl. every gap found in the pilot ports: + * {%comment%}, {%highlight%}, {%raw%}, kramdown attrs, TOC markers, + * {% include X %} inlined from _includes/ (beta-warning → admonition), + * Jekyll literal-escape {{ "{{ x " }}}} → {{ x }}, scalar redirect_from → array. + * 4. Links: strip https://help.keboola.com; remap /cli/ → /cli/keboola-as-code/; + * links to SKIPped pages → their canonical help target; repo-wide inbound + * developers.keboola.com/ links flipped to the ported internal path + * (except files owned by open weave PRs, listed in FLIP_EXCLUDE). + * 5. Images dual-copied (public/ + co-located src/); relative image refs rewritten to + * absolute public paths by basename; non-image assets (csv/zip) copied to public/. + * 6. redirect_from for skipped-but-redirectable dev URLs injected into their canonical + * main pages (guarded, idempotent). + * 7. "Developer Docs" nav group inserted into _data/navigation.yml (guarded) and + * src/sidebar.mjs regenerated. + * 8. MIGRATION-REPORT.md written with every move/skip/redirect/link-flip/inline. + * + * Never prunes. Collisions with existing help slugs are hard errors. + */ +import { + readFileSync, writeFileSync, mkdirSync, readdirSync, statSync, existsSync, copyFileSync, +} from 'node:fs'; +import { join, dirname, extname, relative, basename } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { execSync } from 'node:child_process'; + +const REPO = join(dirname(fileURLToPath(import.meta.url)), '..'); +const DEST_DOCS = join(REPO, 'src', 'content', 'docs'); +const DEST_PUBLIC = join(REPO, 'public'); + +const [, , SRC, ...rest] = process.argv; +const DRY = rest.includes('--dry'); +if (!SRC || !existsSync(SRC)) { + console.error('Usage: node scripts/migrate-devdocs.mjs [--dry]'); + process.exit(1); +} +const INCLUDES = join(SRC, '_includes'); + +const IMG_EXT = new Set(['.png', '.jpg', '.jpeg', '.gif', '.svg', '.webp']); +const ASSET_EXT = new Set(['.csv', '.zip']); + +/* ------------------------------- config ---------------------------------- */ + +// Dev pages NOT ported here: already woven into help by open weave PRs, or dead chrome. +// value = canonical help path links should point at ('' = no link target needed). +const SKIP = new Map(Object.entries({ + '/': '', // weave PR #1022 (combined home) + '/overview/': '/overview/', // weave PR #1022 + '/overview/repositories/': '/overview/', // killed in #1022 (301 there) + '/integrate/data-streams/': '/storage/data-streams/', // weave PR #1023 (301s there) + '/integrate/data-streams/overview/': '/storage/data-streams/', // #1023 → reference/ + '/integrate/data-streams/tutorial/': '/storage/data-streams/', // #1023 → tutorial/ + '/integrate/push-data/': '/storage/data-streams/', // old alias, #1023 + '/integrate/database/': '/components/extractors/database/', // fold PR #1019 + '/integrate/mcp/': '/ai/mcp-server/', // canonical (merged content) + '/integrate/orchestrator/': '/flows/', // empty dev stub + '/404.html': '', // site chrome +})); + +// Skipped dev URLs that get a redirect_from injected into a MAIN canonical page now. +// (data-streams family intentionally absent — its redirects ride with open PR #1023; +// home/overview redirect-less — same path; repositories 301 rides with #1022.) +const INJECT_REDIRECTS = [ + { devUrl: '/integrate/mcp/', canonical: 'ai/mcp-server/index.md' }, + { devUrl: '/integrate/database/', canonical: 'components/extractors/database/index.md' }, + { devUrl: '/integrate/orchestrator/', canonical: 'flows/index.md' }, +]; + +// Slug remap: whole dev sections that land under a different help path. +const REMAP = [ + { from: /^cli(\/|$)/, to: 'cli/keboola-as-code$1', + urlFrom: /^\/cli\//, urlTo: '/cli/keboola-as-code/' }, +]; + +// Scripted banner injected right after the frontmatter of these ported slugs. +const BANNERS = new Map(Object.entries({ + 'cli/keboola-as-code': [ + ':::caution[Deprecated in the future]', + 'The **Keboola as Code CLI** will be deprecated in favor of the new agent-first', + '**[kbagent CLI](https://github.com/keboola/cli)**. Use kbagent for new automation;', + 'this reference stays for existing Keboola-as-Code workflows.', + ':::', + ].join('\n'), +})); + +// Deterministic link corrections for typos in the DEV SOURCE itself (documented, code-reviewed). +const LINK_FIXES = new Map(Object.entries({ + '/cli/keboola-as-code/commands/remote/create/brabch/': '/cli/keboola-as-code/commands/remote/create/branch/', // dev typo "brabch" +})); + +// Repo files NOT touched by the inbound link-flip (owned by open weave PRs — avoid conflicts). +const FLIP_EXCLUDE = [ + 'src/content/docs/index.md', // #1022 + 'src/content/docs/overview/index.md', // #1022 + 'src/content/docs/storage/data-streams/', // #1023 (subtree) + 'src/content/docs/components/extractors/database/index.md', // #1019 +]; + +/* ---------------------------- report plumbing ---------------------------- */ + +const report = { ported: [], skipped: [], collisions: [], includesInlined: [], + bannerInjected: [], redirectsInjected: [], imgFixed: [], assets: [], flips: [], + todos: [], nav: [] }; + +/* ------------------------------ frontmatter ------------------------------ */ + +function parseFm(raw) { + const m = raw.match(/^---\n([\s\S]*?)\n---\n?([\s\S]*)$/); + return m ? { fm: m[1], body: m[2] } : null; +} + +function permalinkToSlug(permalink) { + const v = permalink.trim().replace(/^['"]|['"]$/g, ''); + if (v === '/') return ''; + return v.replace(/^\//, '').replace(/\/$/, ''); +} + +// Line-based: unknown keys pass through; scalar redirect_from normalized to array. +function transformFrontmatter(fmRaw) { + const out = []; + let slug = null; + for (const line of fmRaw.split('\n')) { + const pm = line.match(/^permalink:\s*(.+)$/); + if (pm) { + slug = permalinkToSlug(pm[1]); + for (const r of REMAP) slug = slug.replace(r.from, r.to); + out.push(`slug: '${slug}'`); + continue; + } + const rf = line.match(/^redirect_from:\s*(\S.*)$/); // scalar form breaks Astro schema + if (rf) { + out.push('redirect_from:'); + out.push(` - ${rf[1].trim()}`); + continue; + } + if (/^(layout|showBreadcrumbs):/.test(line)) continue; // Jekyll-only + out.push(line); + } + return { fm: out.join('\n'), slug }; +} + +/* --------------------------------- body ---------------------------------- */ + +const BETA_WARNING_ADMONITION = [ + ':::caution[Public Beta]', + 'This feature is currently in public beta. Please provide feedback using the feedback button in your project.', + ':::', +].join('\n'); + +function inlineIncludes(b, pageRel) { + return b.replace(/\{%\s*include\s+([\w.\-]+)\s*%\}/g, (_, name) => { + if (name === 'branches-beta-warning.html') { + report.includesInlined.push(`${pageRel}: ${name} -> admonition`); + return BETA_WARNING_ADMONITION; + } + const f = join(INCLUDES, name); + if (!existsSync(f)) { + report.todos.push(`${pageRel}: include ${name} NOT FOUND — left as TODO`); + return ``; + } + const content = readFileSync(f, 'utf8').replace(/\r\n/g, '\n').trimEnd(); + if (/\.(html)$/.test(name)) { + report.todos.push(`${pageRel}: raw HTML include ${name} inlined — verify rendering`); + } else { + report.includesInlined.push(`${pageRel}: ${name} inlined (${content.split('\n').length} lines)`); + } + return content; + }); +} + +// jQuery/style page furniture that only existed for the Jekyll site. +const stripPageScripts = (b) => b + .replace(/ + diff --git a/src/content/docs/extend/generic-extractor/configuration/iterations/index.md b/src/content/docs/extend/generic-extractor/configuration/iterations/index.md new file mode 100644 index 000000000..a2894b7d0 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/configuration/iterations/index.md @@ -0,0 +1,258 @@ +--- +title: Iterations +slug: 'extend/generic-extractor/configuration/iterations' +redirect_from: + - /extend/generic-extractor/iterations/ +--- + + +The `iterations` section allows you to **execute a configuration multiple times, each time with different +values**. The most typical use for `iterations` is extraction of the same data from multiple accounts. +Iterations can always be replaced by creating multiple complete configurations of Generic Extractor. + +Iterations are specified as an array of objects, where each object contains the same properties +as the [`config`](/extend/generic-extractor/configuration/config/) section. All properties of the object are optional. + +Consider the following example of an `iterations` configuration defining that the entire Generic Configuration +will be executed twice: the first time with the username `JohnDoe`, and the second time with the username +`DoeJohn`. + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/", + "authentication": { + "type": "basic" + } + }, + "config": { + "outputBucket": "ge-tutorial", + "jobs": [ + { + "endpoint": "users" + } + ] + }, + "iterations": [ + { + "username": "JohnDoe", + "#password": "TopSecret" + }, + { + "username": "DoeJohn", + "#password": "EvenMoreSecret" + } + ] + } +} +``` + +Since **all `iterations` properties override the `config` properties**, they are accessible +as [configuration attributes](/extend/generic-extractor/functions/#configuration-attributes) +via the `attr` property. + +Keep in mind that `iterations` can refer directly only to the things specified in the `config` section. +For the `api` section, you must use [functions](/extend/generic-extractor/functions/). +Also, it is not possible to iterate over values returned in the response. +The number of iterations and their values must be defined in the configuration. + +## Configuration +Because the values defined in `iterations` override those in the `config` section, +everything that can be in the `config` section is allowed as well +(including arbitrary user attributes used in [functions](/extend/generic-extractor/functions/)). +Using `jobs` and `mappings` in iterations does not make much sense though. + +Also, if you use `userData` in iterations, they must result in the same columns; otherwise the resulting +table cannot be imported into Storage. If you use `incrementalOutput`, only the last value of `incrementalOutput` +is honoured. + +## Examples + +### Iterating Parameters +Suppose, you have an API which takes a URL parameter `account_id`, which restricts the returned data to a +certain account. The following configuration executes the entire configuration for two accounts --- `345` and `456`: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/" + }, + "config": { + "outputBucket": "ge-tutorial", + "userData": { + "account": { + "attr": "accountId" + } + }, + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "account_id": { + "attr": "accountId" + } + } + } + ] + }, + "iterations": [ + { + "accountId": 345 + }, + { + "accountId": 456 + } + ] + } +} +``` + +Since the `iterations` section overrides the values in the `config` section, the below configuration +yields the exact same results as the configuration above: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/" + }, + "config": { + "outputBucket": "ge-tutorial", + "accountId": 123, + "userData": { + "account": { + "attr": "accountId" + } + }, + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "account_id": { + "attr": "accountId" + } + } + } + ] + }, + "iterations": [ + { + "accountId": 345 + }, + { + "accountId": 456 + } + ] + } +} +``` + +It looks as if the first execution is with `account_id=123`, but it is not the case. The configuration +will be executed only twice: the first time with `account_id=345` and the second time with `account_id=456`. +See [example [EX112]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/112-iterations-params). + +### Iterating Headers +Suppose you have an API from which you want to extract data from two accounts (`JohnDoe` and `DoeJohn`). The +API uses the [HTTP Basic Authentication](/extend/generic-extractor/configuration/api/authentication/basic/) method, and in addition, +each user has their own API token, which must be provided in the `X-Api-Token` header. + +Even if the above parameters relate to the [`api` configuration](/extend/generic-extractor/configuration/api/), which cannot +be directly included in `iterations`, we can specify them as `http.headers` and therefore it is still possible to use them. + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/", + "authentication": { + "type": "basic" + } + }, + "config": { + "outputBucket": "ge-tutorial", + "jobs": [ + { + "endpoint": "users", + "dataType": "users" + } + ] + }, + "iterations": [ + { + "http": { + "headers": { + "X-Api-Token": "1234abcd" + } + }, + "username": "JohnDoe", + "#password": "TopSecret" + }, + { + "http": { + "headers": { + "X-Api-Token": "zyxv9876" + } + }, + "username": "DoeJohn", + "#password": "EvenMoreSecret" + } + ] + } +} +``` + +Next to `username` and `#password` from the `config` section, the above configuration overrides also +the `http.headers.X-Api-Token` setting. The configuration can be simplified by using +[functions and references](/extend/generic-extractor/functions/): + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/", + "authentication": { + "type": "basic" + } + }, + "config": { + "http": { + "headers": { + "X-Api-Token": { + "attr": "apiToken" + } + } + }, + "outputBucket": "ge-tutorial", + "jobs": [ + { + "endpoint": "users", + "dataType": "users" + } + ] + }, + "iterations": [ + { + "apiToken": "1234abcd", + "username": "JohnDoe", + "#password": "TopSecret" + }, + { + "apiToken": "zyxv9876", + "username": "DoeJohn", + "#password": "EvenMoreSecret" + } + ] + } +} +``` + +Here, the `config` section specifies the part of the token authentication which is common to both iterations. +Each iteration then specifies only the token. By writing `"attr": "apiToken"` under `X-Api-Token` we say that +`X-Api-Token` will have the value of the `apiToken` property from the `config` section. That property is in +turn specified in the iteration objects. + +See [example [EX113]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/113-iterations-headers). diff --git a/src/content/docs/extend/generic-extractor/configuration/ssh-proxy/index.md b/src/content/docs/extend/generic-extractor/configuration/ssh-proxy/index.md new file mode 100644 index 000000000..c6a2cdb9b --- /dev/null +++ b/src/content/docs/extend/generic-extractor/configuration/ssh-proxy/index.md @@ -0,0 +1,91 @@ +--- +title: SSH Proxy Configuration +slug: 'extend/generic-extractor/configuration/ssh-proxy' +--- + + +*To configure your first Generic Extractor, follow our [tutorial](/extend/generic-extractor/tutorial/).* +*Use [Parameter Map](/extend/generic-extractor/map/) to help you navigate among various +configuration options.* + +An SSH proxy for Generic Extractor allows you tu securely access HTTP(s) endpoints inside your private network. +It creates an SSH tunnel, and all traffic from Generic Extractor is forwarded through the tunnel to the destination server. + +A sample `config` configuration can look like this: + +```json +{ + ..., + "sshProxy": { + "host": "proxy.example.com", + "user": "proxy", + "port": 22, + "#privateKey": "-----BEGIN RSA PRIVATE KEY-----\n...\n-----END RSA PRIVATE KEY-----" + } +} +``` + +## Usage +Before using an SSH proxy, set up an **SSH proxy server** +to act as a gateway to your private network where your destination server resides. + +Complete the following steps to set up an SSH proxy for Generic Extractor: + +### 1. Set Up SSH Proxy Server +Here is a very basic [Dockerfile](https://docs.docker.com/engine/reference/builder/) example. +All it does is run an sshd daemon and expose port 22. You can, of course, set this up in your system in +a similar way without using Docker. + +```dockerfile +FROM ubuntu:14.04 + +RUN apt-get update + +RUN apt-get install -y openssh-server +RUN mkdir /var/run/sshd + +RUN echo 'root:root' |chpasswd + +RUN sed -ri 's/^PermitRootLogin\s+.*/PermitRootLogin yes/' /etc/ssh/sshd_config +RUN sed -ri 's/UsePAM yes/#UsePAM yes/g' /etc/ssh/sshd_config + +EXPOSE 22 + +CMD ["/usr/sbin/sshd", "-D"] +``` + +This server should be in the same private network where your destination server resides. It should be accessible publicly from the internet via SSH. +The default port for SSH is 22, but you can choose a different port. + +We highly recommend to allow access only from the [Keboola IP address ranges](/extractors/ip-addresses/). + +See the following pages for more information about setting up SSH on your server: + +- [OpenSSH configuration](https://help.ubuntu.com/community/SSH/OpenSSH/Configuring) +- [Dockerized SSH service](https://docs.docker.com/engine/examples/running_ssh_service/) + +### 2. Generate SSH Key Pair +Generate an SSH key pair and copy the public key to your **SSH proxy server**. +Paste it to the **public.key** file, and then append it to the authorized_keys file. + +```bash +mkdir ~/.ssh +cat public.key >> ~/.ssh/authorized_keys +``` + +### 3. Configure Generic Extractor SSH Proxy + +```json +{ + ..., + "sshProxy": { + "host": "your-ssh-proxy-host", + "user": "ssh-proxy-user", + "port": 22, + "#privateKey": "your-generated-private-key" + } +} +``` + +See [example [EX131]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/131-ssh-tunnel). +and [example [EX133]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/133-ssh-tunnel-iterations-params). diff --git a/src/content/docs/extend/generic-extractor/configuration/ui_switch.png b/src/content/docs/extend/generic-extractor/configuration/ui_switch.png new file mode 100644 index 000000000..83152c49a Binary files /dev/null and b/src/content/docs/extend/generic-extractor/configuration/ui_switch.png differ diff --git a/src/content/docs/extend/generic-extractor/events.png b/src/content/docs/extend/generic-extractor/events.png new file mode 100644 index 000000000..3609158ca Binary files /dev/null and b/src/content/docs/extend/generic-extractor/events.png differ diff --git a/src/content/docs/extend/generic-extractor/function_eval.gif b/src/content/docs/extend/generic-extractor/function_eval.gif new file mode 100644 index 000000000..4513c09cc Binary files /dev/null and b/src/content/docs/extend/generic-extractor/function_eval.gif differ diff --git a/src/content/docs/extend/generic-extractor/functions.png b/src/content/docs/extend/generic-extractor/functions.png new file mode 100644 index 000000000..0a893a9f2 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/functions.png differ diff --git a/src/content/docs/extend/generic-extractor/functions/index.md b/src/content/docs/extend/generic-extractor/functions/index.md new file mode 100644 index 000000000..6087f1b90 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/functions/index.md @@ -0,0 +1,1504 @@ +--- +title: Functions +slug: 'extend/generic-extractor/functions' +--- + + +Functions are simple pre-defined functions that + +- allow you to add extra flexibility when needed. +- can be used in several places in the Generic Extractor configuration to introduce dynamically generated values instead of +those provided statically. +- allow referencing the existing values in the configuration instead of copying them. +- are advantageous and sometimes necessary when [publishing your configuration as a new component](/extend/generic-extractor/publish/). + +## Configuration +A function is used instead of a simple value in specific parts of the Generic Extractor configuration (see [below](#function-contexts)). +A function configuration is an object with the properties `function` (one of the [available function names](#supported-functions) and `args` +(function arguments), for example: + +```json +{ + "function": "concat", + "args": [ + "John", + "Doe" + ] +} +``` + +The argument of a function can be any of the following: + +- [Scalar](/extend/generic-extractor/tutorial/json/#data-values) (simple) value (as in the above example) +- Reference to a value from [function context (see below)](#function-contexts) +- Another function object + +Additionally, the function may be replaced by a plain reference to the function context. This means you can write (where permitted) +a configuration value in three possible ways: + +**A simple value:** + +```json +{ + ..., + "baseUrl": "http://example.com/ +} +``` + +**A function call:** + +```json +{ + ..., + "baseUrl": { + "function": "concat", + "args": [ + "http://", + "example.com" + ] + } +} +``` + +**A reference to a value from the function context:** +```json +{ + ..., + "baseUrl": { + "attr": "someUrl" + } +} +``` + +These forms can be combined freely. They can be also nested in a virtually unlimited way. For instance: + +```json +{ + ..., + "baseUrl": { + "function": "concat", + "args": [ + "https://", + { + "attr": "domain" + } + ] + } +} +``` + +### User Interface +You may create functions in the user interface's `User Parameters` or `User Data` sections. +You can also create the functions directly from other configuration contexts, e.g., when defining the query parameters on the endpoint. + +Aside from predefined functions, the UI also offers the most common templates that you can use. + +![img.png](/extend/generic-extractor/functions.png) + +The UI also offers a convenient way to evaluate the function and see the results. + +![img.png](/extend/generic-extractor/function_eval.gif) + +## Supported Functions + +### md5 +The [`md5` function](https://www.php.net/manual/en/function.md5.php) calculates the [MD5 hash](https://en.wikipedia.org/wiki/MD5) of a +string. The function takes one argument, which is the string to hash. + +```json +{ + "function": "md5", + "args": [ + "NotSoSecret" + ] +} +``` + +The above will produce `1228d3ff5089f27721f1e0403ad86e73`. + +See an [example](#job-parameters). + +### sha1 +The [`sha1` function](https://www.php.net/manual/en/function.sha1.php) calculates the [SHA-1 hash](https://en.wikipedia.org/wiki/SHA-1) of a +string. The function takes one argument which is the string to hash. + +```json +{ + "function": "sha1", + "args": [ + "NotSoSecret" + ] +} +``` + +The above will produce `64d5d2977cc2573afbd187ff5e71d1529fd7f6d8`. + +See an [example](#job-parameters). + +### base64_encode +The [`base64_encode` function](https://www.php.net/manual/en/function.base64-encode.php) converts a +string to the [MIME Base64 encoding](https://en.wikipedia.org/wiki/Base64#MIME). The function +takes one argument which is the string to encode. + +```json +{ + "function": "base64_encode", + "args": [ + "TeaPot" + ] +} +``` + +The above will produce `VGVhUG90`. + +See an [example](#nested-functions). + +### hash_hmac +The [`hash_hmac` function](https://www.php.net/manual/en/function.hash-hmac.php) creates +an [HMAC (Hash-based message authentication code)](https://en.wikipedia.org/wiki/Hash-based_message_authentication_code) +from a string. The function takes +three arguments: + +1. Name of a hashing algorithm (see the +[list of supported algorithms](https://www.php.net/manual/en/function.hash-algos.php#refsect1-function.hash-algos-examples)) +2. Value to hash +3. Secret key + +```json +{ + "function": "hash_hmac", + "args": [ + "sha256", + "12345abcd5678efgh90ijk", + "TeaPot" + ] +} +``` + +The above will return `d868d581b2f2edd09e8e7ce12c00723b3fcffb6a5d74c40eae9d94181a0bf731`. + +See an [example](#api-default-parameters). + +### hash +This function works similarly to the `hash_hmac` function but requires only two arguments (no secret key required): + +1. The name of a hashing algorithm (see the + [list of supported algorithms](https://www.php.net/manual/en/function.hash-algos.php#refsect1-function.hash-algos-examples)). +2. The value to hash. + +```json +{ + "function": "hash", + "args": [ + "sha256", + "12345abcd5678efgh90ijk" + ] +} +``` + +### time +The [`time` function](https://www.php.net/manual/en/function.time.php) returns the current time as a +[Unix timestamp](https://en.wikipedia.org/wiki/Unix_time). +To obtain the current time in a more readable format, use the +the [`date` function](#date). It takes no arguments. + +```json +{ + "function": "time" +} +``` + +The above will produce something like `1492674974`. + +### date +The [`date` function](https://www.php.net/manual/en/function.date.php) formats the provided or the current +timestamp into a human readable format. The function takes either one or two arguments: + +1. [Formatting string](https://www.php.net/manual/en/function.date.php#refsect1-function.date-parameters) +2. Optional [Unix timestamp](https://en.wikipedia.org/wiki/Unix_time); if not provided, the current time is used. + +```json +{ + "function": "date", + "args": [ + "Y-m-d" + ] +} +``` + +The above will produce something like `2017-04-20`. + +```json +{ + "function": "date", + "args": [ + "Y-m-d H:i:s", + 1490000000 + ] +} +``` + +The above will produce `2017-03-20 8:53:20`. + +See an [example](#user-data). + +### strtotime +The [`strtotime` function](https://www.php.net/manual/en/function.strtotime.php) converts a string date into a [Unix timestamp](https://en.wikipedia.org/wiki/Unix_time). The function takes +one or two arguments: + +1. String date +2. Base for relative dates (see below) + +```json +{ + "function": "strtotime", + "args": [ + "21 oct 2017 9:16pm" + ] +} +``` + +The above will produce `1508620560`, which represents the date `2017-10-21 21:16:00`. However, the +[`strtotime` function](https://www.php.net/manual/en/function.strtotime.php) is most useful with relative dates which it also allows. For example, you can +write: + +```json +{ + "function": "strtotime", + "args": [ + "-7 days", + 1508620560 + ] +} +``` + +The above will give `1508015760`, which represents the date `2017-10-14 21:16:00`. The second argument +specifies the base date (as a Unix timestamp) from which the relative date is computed. This is particularly +useful for [incremental extraction](/extend/generic-extractor/incremental/). Also note that +it is common to combine the `strtotime` and `date` functions to convert between string and timestamp +representation of a date. + +See an [example](#nested-strtotime). + +### sprintf +The `sprintf` function formats values and inserts them into a string. The `sprintf` function maps directly to +the [original PHP function](https://www.php.net/manual/en/function.sprintf.php), which is very versatile and has many +uses. The function accepts two or more arguments: + +1. String with [formatting directives](https://www.php.net/manual/en/function.sprintf.php) (marked with the percent character `%`) +2. Values inserted into the string: + +```json +{ + "function": "sprintf", + "args": [ + "Three %s are %.2f %s.", + "apples", + 0.5, + "plums" + ] +} +``` + +The above will produce `Three apples are 0.50 plums.` + +See a [simple insert example](#api-base-url) or a [formatting example](#job-placeholders). + +### concat +The `concat` function concatenates an arbitrary number of strings into one. For example: + +```json +{ + "function": "concat", + "args": [ + "Hen", + "Or", + "Egg" + ] +} +``` + +The above will produce `HenOrEgg` (see [example 1](#api-base-url), [example 2](#headers)). See also the +[`implode` function](#implode). + +### implode +The [`implode` function](https://www.php.net/manual/en/function.implode.php) concatenates an arbitrary number +of strings into one using a delimiter. The function takes +two arguments: + +1. Delimiter string which is used for the concatenation +2. Array of values to be concatenated + +For example: + +```json +{ + "function": "implode", + "args": [ + ",", + [ + "apples", + "oranges", + "plums" + ] + ] +} +``` + +The above will produce `apples,oranges,plums` (see an [example](#headers)). +The delimiter can be empty, in which case the `implode` function is equivalent to the [`concat` function](#concat): + +```json +{ + "function": "implode", + "args": [ + "", + [ + "Hen", + "Or", + "Egg" + ] + ] +} +``` + +### ifempty +The `ifempty` function can be useful for handling optional values. The function takes two arguments and +returns the first one if it is not empty. If the first argument is empty, it returns the second argument. + +```json +{ + "function": "ifempty", + "args": [ + "", + "Banzai" + ] +} +``` + +The above will return `Banzai`. For the `ifempty` function, an empty string and the values `0` and `null` are +considered 'empty'. + +See an [example](#optional-job-parameters). + +## Function Contexts +Every place in the Generic Extractor configuration in which a function may be used may allow different arguments of the function. +This is referred to as a **function context**. Many contexts share access to **configuration attributes**. + +### Configuration Attributes +The configuration attributes are accessible in specific function contexts and they represent the entire [`config`](/extend/generic-extractor/configuration/config/) +section of the Generic Extractor configuration. There is some processing involved: + +- The [`jobs`](/extend/generic-extractor/configuration/config/jobs/) section is removed entirely. +- All other values are flattened (keys are concatenated using a dot `.`) into a one-level deep object. +- The result object is available in a property named `attr`. + +For example, the following configuration: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com" + }, + "config": { + "debug": true, + "outputBucket": "get-tutorial", + "server": "localhost:8888", + "incrementalOutput": false, + "jobs": [ + { + "endpoint": "users", + "dataType": "users" + } + ], + "http": { + "headers": { + "X-AppKey": "ThisIsSecret", + "X-Auth": { + "function": "concat", + "args": [ + "Tea", + "Pot" + ] + } + } + }, + "userData": { + "tag": "fullExtract", + "mode": "development" + }, + "mappings": { + "content": { + "whatever": "foobar" + } + } + } + } +} +``` + +will be converted to the following function context: + +```json +{ + "attr": { + "debug": true, + "outputBucket": "mock-server", + "server": "localhost:8888", + "incrementalOutput": false, + "http.headers.X-AppKey": "ThisIsSecret", + "http.headers.X-Auth.function": "concat", + "http.headers.X-Auth.args.0": "Tea", + "http.headers.X-Auth.args.1": "Pot", + "userData.tag": "fullExtract", + "userData.mode": "development", + "mappings.content.whatever": "foobar" + } +} +``` + +See [example [EX119]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/119-function-nested-config). + +### Base URL Context +The Base URL function context is used when setting the [`baseURL` for API](/extend/generic-extractor/configuration/api/#base-url), and it +contains [configuration attributes](/#function-contexts). + +See an [example](#api-base-url). + +### Headers Context +The Headers function context is used when setting the [`http.headers` for API](/extend/generic-extractor/configuration/api/#headers) +or the [`http.headers` in config](/extend/generic-extractor/configuration/config/#http), and it contains +[configuration attributes](/#function-contexts). + +See an [example](#headers). + +### Parameters Context +The Parameters function context is used when setting job [request parameters --- `params`](/extend/generic-extractor/configuration/config/jobs/#request-parameters). +It contains [configuration attributes](/#function-contexts) plus the times of the current +(`currentStart`) and previous (`previousStart`) run of Generic Extractor. +The times are [Unix timestamps](https://en.wikipedia.org/wiki/Unix_time). +If the extraction is run for the first time, `previousStart` is 0. + +With the following configuration: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com" + }, + "config": { + "debug": true, + "outputBucket": "get-tutorial", + "server": "localhost:8888", + "jobs": [ + ... + ] + } + } +} +``` + +the parameters function context will contain: + +```json +{ + "attr": { + "debug": true, + "outputBucket": "mock-server", + "server": "localhost:8888" + }, + "time": { + "previousStart": 0, + "currentStart": 1492678268 + } +} +``` + +See an [example of using parameters context](#job-parameters). + +The `time` values are used in [incremental processing](/extend/generic-extractor/incremental/). + +### Placeholder Context +The Placeholder function context refers to configuration of [placeholders in child jobs](/extend/generic-extractor/configuration/config/jobs/children/#placeholders). +When using function to process a placeholder value, the placeholder must be specified as an object with the `path` property. +Therefore instead of writing: + +```json +"placeholders": { + "user-id": "userId" +} +``` + +write: + +```json +"placeholders": { + "user-id": { + "path": "userId", + "function": ... + } +} +``` + +The placeholder function context contains the following structure: + +```json +{ + "placeholder": { + "value": "???" + } +} +``` + +where `???` is the value obtained from the response JSON from the path provided in the `path` property +of the placeholder. + +See an [example](#job-placeholders). + +### User Data Context +The User Data function context is used when setting the [`userData`](/extend/generic-extractor/configuration/config/#user-data). +The parameters context contains [configuration attributes](/#function-contexts) plus the times of the current (`currentStart`) and +previous (`previousStart`) run of Generic Extractor. The User Data Context is therefore +same as the [Parameters Context](#parameters-context). + +See an [example](#user-data). + +### Login Authentication Context +The Login Authentication function context is used in the +[login authentication](/extend/generic-extractor/configuration/api/authentication/login/) method. +Functions are supported in both [`loginRequest`](/extend/generic-extractor/configuration/api/authentication/login/#configuration-parameters) +and [`apiRequest` ](/extend/generic-extractor/configuration/api/authentication/login/#configuration-parameters) configurations. +The `loginRequest` function context contains [configuration attributes](/#function-contexts). +In the `apiRequest` context, the flattened reponse of the login request is available additionally +to the [configuration attributes](/#function-contexts). +The login authentication context is the same for both `params` and `headers` +[login authentication configuration options](/extend/generic-extractor/configuration/api/authentication/login/#configuration-parameters). If the +login authentication request returns e.g.: + +```json +{ + "user": "John Doe", + "authorization": { + "token": "quiteSecret", + "validUntil": "2017-20-12 12:20:17" + } +} +``` + +The following function context will be available in the API request headers and query: + +```json +{ + "attr": { + "outputBucket": "mock-server" + }, + "response": { + "user": "John Doe", + "authorization.token": "quiteSecret", + "authorization.validUntil": "2017-20-12 12:20:17" + } +} +``` + +The login response is available in the `response` node. The `attr` node contains [configuration attributes](/extend/generic-extractor/functions/#configuration-attributes). +See an [example](/extend/generic-extractor/configuration/api/authentication/login/#login-authentication-with-functions) and a more +[complicated example](/extend/generic-extractor/configuration/api/authentication/login/#login-authentication-with-login-and-api-request) of using functions in +both login request and API request. + +### Query Authentication Context +The Query Authentication function context is used in the +[query authentication](/extend/generic-extractor/configuration/api/authentication/query/) method. +The Query Authentication Context contains [configuration attributes](/#function-contexts) plus +a representation of the complete HTTP request to be sent (`request`) plus a key +value list of query parameters of the HTTP request (`query`). + +The following configuration: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/", + "http": { + "defaultOptions": { + "params": { + "account": "admin" + } + } + }, + "authentication": { + "type": "query", + "query": { + "signature": { + "function": "sha1", + "args": [ + "time", + { + "attr": "#api-key" + } + ] + } + } + } + }, + "config": { + "#api-key": "12345abcd5678efgh90ijk", + "outputBucket": "mock-server", + "jobs": [ + { + "endpoint": "users", + "params": { + "showColumns": "all" + } + } + ] + } + } +} +``` + +leads to the following function context: + +```json +{ + "query": { + "account": "admin", + "showColumns": "all" + }, + "request": { + "url": "http:\/\/example.com\/users?account=admin&showColumns=all", + "path": "\/users", + "queryString": "account=admin&showColumns=all", + "method": "GET", + "hostname": "example.com", + "port": 80, + "resource": "\/users?account=admin&showColumns=all" + }, + "attr": { + "#api-key": "12345abcd5678efgh90ijk", + "outputBucket": "mock-server" + } +} +``` + +See the [basic example](#api-default-parameter) and a [more complicated example](#api-query-authentication). + +### OAuth 2.0 Authentication Context +The OAuth Authentication Context is used for the +[`oauth20`](/extend/generic-extractor/configuration/api/authentication/oauth20/) authentication method +(it is not applicable to `oauth10`) and contains the following: + +- Representation of the complete HTTP request to be sent (`request`) +- A key value list of query parameters of the HTTP request (`query`) +- An `authorization` section containing the response from the OAuth service provider + +This context is available for both the `headers` and `query` sections of the `oauth20` authentication methods. + +The following configuration: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/", + "authentication": { + "type": "oauth20", + "format": "json", + "headers": { + "Authorization": { + "function": "concat", + "args": [ + "Bearer ", + { + "authorization": "#data.access_token" + } + ] + } + } + } + }, + "config": { + "outputBucket": "mock-server", + "jobs": [ + { + "endpoint": "users", + "dataType": "users" + } + ] + } + }, + "authorization": { + "oauth_api": { + "credentials": { + "#data": "{\"status\": \"ok\",\"access_token\": \"testToken\", \"foo\": {\"bar\": \"baz\"}}", + "appKey": "clientId", + "#appSecret": "clientSecret" + } + } + } +} +``` + +leads to the following function context: + +```json +{ + "query": { + "showColumns": "all" + }, + "request": { + "url": "http:\/\/example.com\/users?showColumns=all", + "path": "\/users", + "queryString": "showColumns=all", + "method": "GET", + "hostname": "example.com", + "port": 80, + "resource": "\/users?showColumns=all" + }, + "authorization": { + "data.status": "ok", + "data.access_token": "testToken", + "data.foo.bar": "baz" + "timestamp": 1492949837, + "nonce": "99206d94a6846841", + "clientId": "clientId", + } +} +``` + +The `authorization` section of the configuration contains the +[OAuth2 response](/extend/generic-extractor/configuration/api/authentication/oauth20/). The function context contains +the parsed and flattened response fields under the key `data`, provided that the response was sent in JSON format +and that [`"format": "json"`](/extend/generic-extractor/configuration/api/authentication/oauth20/#configuration) was set. + +In the response above, these are the keys `data.status`, `data.access_token`, `data.foo.bar`. This is defined +entirely by the behavior of the OAuth Service provider. If the response is a plaintext (usually directly a token), +then the entire response is available in the field `data`. + +Apart from that, the fields `timestamp` (Unix timestamp of the request), +`nonce` (cryptographic [nonce](https://en.wikipedia.org/wiki/Cryptographic_nonce) for +signing the request) and `clientId` (the value of `authorization.oauth_api.credentials.appKey`, which is obtained when +the application is published) are added to the `authorization` section. + +For usage, see [OAuth examples](/extend/generic-extractor/configuration/api/authentication/oauth20/). + +### OAuth 2.0 Login Authentication Context +The OAuth Login Authentication Context is used for the +[`oauth20.login`](/extend/generic-extractor/configuration/api/authentication/oauth20-login/) authentication method +(it is not applicable to `oauth20`). The OAuth Login Authentication context contains +OAuth information split into the properties `consumer` (response obtained from the service provider) and +`user` (data obtained from the user). This context is available for +both the `headers` and `params` sections of the `oauth20` authentication methods. +For the context available in the `apiRequest` configuration, see the [login authentication](/extend/generic-extractor/functions/#login-authentication-context). + +The following configuration: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com", + "authentication": { + "type": "oauth20.login", + ... + } + }, + "config": { + ... + } + }, + "authorization": { + "oauth_api": { + "credentials": { + "#data": "{\"status\": \"ok\",\"access_token\": \"testToken\", \"mac_secret\": \"iAreSoSecret123\", \"foo\": {\"bar\": \"baz\"}}", + "appKey": "clientId", + "#appSecret": "clientSecret" + } + } + } +} +``` + +leads to the following function context: + +```json +{ + "consumer": { + "client_id": "clientId", + "client_secret": "clientSecret" + }, + "user": { + "status": "ok", + "access_token": "testToken", + "mac_secret": "iAreSoSecret123", + "foo.bar": "baz" + } +} +``` + +The `authorization` section of the configuration contains the +[OAuth2 response](/extend/generic-extractor/configuration/api/authentication/oauth20/). The function context +contains the parsed and flattened response fields in the `user` property. The content of the +`user` property is fully dependent on the response of the OAuth service provider. The +`consumer` property contains the `client_id` and `client_secret` which contain values of +`authorization.oauth_api.credetials.appKey` and +`authorization.oauth_api.credetials.appSecret` respectively. +(These are obtained by Keboola when the application is published). + +For usage, see [OAuth Login examples](/extend/generic-extractor/configuration/api/authentication/oauth20-login/). + +## Examples + +### API Base URL +When [publishing your Generic Extractor configuration](/extend/generic-extractor/publish/), chances are +you want the end-user to provide a part of the API configuration. Due to the limitations of +[how templates work](/extend/generic-extractor/publish/#configuration-considerations), the parameter +obtained from the end-user configuration will be only available in the `config` section. + +Let's say that the end-user enters `www.example.com` as the API server and that values become +available as the `server` property of the `config` section, for instance: + +```json +"config": { + "outputBucket": "ge-tutorial", + "server": "www.example.com", + "jobs": [ + { + "endpoint": "users", + "dataType": "users" + } + ] +} +``` + +This means that the [configuration attributes](#configuration-attributes) will be available as: + +```json +{ + "attr": { + "outputBucket": "ge-tutorial", + "server": "www.example.com" + } +} +``` + +Then use the [`concat` function](#concat) to access that value and merge it with other parts to create the +final API URL (`http://example.com/api/1.0/`): + +```json +{ + "parameters": { + "api": { + "baseUrl": { + "function": "concat", + "args": [ + "http://", + { + "attr": "server" + }, + "/api/1.0/" + ] + } + } + } +} +``` + +See [example [EX087] with concat](https://github.com/keboola/generic-extractor/tree/master/doc/examples/087-function-baseurl) +or an alternative [example [EX088] with sprintf](https://github.com/keboola/generic-extractor/tree/master/doc/examples/088-function-baseurl-sprintf). + +### API Default Parameters +Suppose you have an API which expects a `tokenHash` parameter to be sent with every request. The +token hash is supposed to be generated by the SHA-256 hashing algorithm from a token and secret +you obtain. + +Because the [`api.http.defaultOptions.params`](/extend/generic-extractor/configuration/api/#headers) option does not +support functions, either supply the parameters in the [`jobs.params`](/extend/generic-extractor/configuration/config/jobs/#request-parameters) +configuration, or use [API Query Authentication](/extend/generic-extractor/configuration/api/authentication/query/). +Using (or abusing) the API Query Authentication is possible if the default parameters represent authentication, or +if the API does not use any authentication method (two authentication methods are not possible): + +The below configuration reads the `#api-key` and `#secret-key` parameters from the `config` section, +computes SHA-256 hash and sends it as a `tokenHash` parameter with every request. + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/", + "authentication": { + "type": "query", + "query": { + "tokenHash": { + "function": "hash_hmac", + "args": [ + "sha256", + { + "attr": "#api-key" + }, + { + "attr": "#secret-key" + } + ] + } + } + } + }, + "config": { + "#api-key": "12345abcd5678efgh90ijk", + "#secret-key": "TeaPot", + "debug": true, + "outputBucket": "mock-server", + "jobs": [ + { + "endpoint": "users", + "dataType": "users" + } + ] + } + } +} +``` + +See [example [EX099]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/099-function-query-parameters). + +The solution with using the `jobs.params` configuration can look like this: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/" + }, + "config": { + "#api-key": "12345abcd5678efgh90ijk", + "#secret-key": "TeaPot", + "debug": true, + "outputBucket": "mock-server", + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "tokenHash": { + "function": "hash_hmac", + "args": [ + "sha256", + { + "attr": "#api-key" + }, + { + "attr": "#secret-key" + } + ] + } + } + } + ] + } + } +} +``` + +The only practical difference is that the `tokenHash` parameter is going to be sent only with +the single `users` job. + +See [example [EX098]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/098-function-hmac). + +### API Query Authentication +Suppose you have an API with only a single endpoint `/items` to which you have to +pass a `type` parameter to list resources of a given type. On top of that, the API requires +an `apiToken` parameter and a `signature` parameter (a hash of the token and type) to be sent with every request. +The following configuration handles the situation: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://mock-server:80/101-function-query-auth/", + "authentication": { + "type": "query", + "query": { + "apiToken": { + "attr": "#token" + }, + "signature": { + "function": "sha1", + "args": [ + { + "function": "concat", + "args": [ + { + "attr": "#token" + }, + { + "query": "type" + } + ] + } + ] + } + }, + "apiRequest": { + "headers": { + "X-Api-Token": "token" + } + } + } + }, + "config": { + "#token": "1234abcd567efg890hij", + "debug": true, + "outputBucket": "mock-server", + "jobs": [ + { + "endpoint": "items", + "dataType": "users", + "params": { + "type": "users" + } + }, + { + "endpoint": "items", + "dataType": "orders", + "params": { + "type": "orders" + } + } + ] + } + } +} +``` + +There are two jobs, both to the same endpoint (`items`), but with a different `type` parameter and `dataType`. +The authentication method `query` adds two more parameters to each request: `apiToken` (contain the value +of `config.#token`) and `signature`. The `signature` parameter is created as an SHA-1 hash of the +token and resource type (`"query": "type"` is taken from the `jobs.params.type` value). + +See [example [EX101]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/101-function-query-auth). + +### Job Placeholders +Let's say you have an API with an endpoint `/users`, returning a list of users, and an +endpoint `/user/{userId}`, returning details of a specific user with a given ID. The list response +looks like this: + +```json +[ + { + "id": 3, + "name": "John Doe" + }, + { + "id": 234, + "name": "Jane Doe" + } +] +``` + +To obtain the details of the first user, the user-id has to be padded to five digits. The details API call for the +first user must be sent to `/user/00003`, and for the second user to `/user/00234`. To achieve this, use the +`sprintf` function, which allows [number padding](https://www.php.net/manual/en/function.sprintf.php#example-6129). + +The following `placeholders` configuration in the child job calls the function with the first argument set to +`%'.05d` (which is a sprintf [format](https://www.php.net/manual/en/function.sprintf.php) to pad with zero to five digits) +and the second argument set to the value of the `id` property found in the parent response. The placeholder path must +be specified in the `path` property. That means that the configuration: + +```json +"placeholders": { + "user-id": "id" +} +``` + +has to be converted to: + +```json +"placeholders": { + "user-id": { + "path": "id", + "function": "sprintf", + "args": [ + "%'.05d", + { + "placeholder": "value" + } + ] + } +} +``` + +The following `user-detail` table will be extracted: + +|id|name|address\_city|address\_country|address\_street|parent\_id| +|123|John Doe|London|UK|Whitehaven Mansions|00003| +|234|Jane Doe|St Mary Mead|UK|High Street|00234| + +Notice that the `parent_id` column contains the processed value and not the original one. + +See [example [EX085]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/085-function-job-placeholders), +or a not-so-useful [example [EX086]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/086-function-job-placeholders-reference) +(using reference). + +### Job Parameters +Let's say you have an API which requires you to send a hash of a certain value with every request. Specifically, +each request must be done with the [HTTP POST method](/extend/generic-extractor/tutorial/rest/#method) with content: + +```json +{ + "token": "someValue" +} +``` + +The following configuration does exactly that. The value of the token is taken from the configuration +root (using the `attr` reference). This is useful in case the configuration is used as part of a +[template](/extend/generic-extractor/publish/). The actual hash will be generated of the `NotSoSecret` value. + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/" + }, + "config": { + "debug": true, + "outputBucket": "mock-server", + "tokenValue": "NotSoSecret", + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "method": "POST", + "params": { + "token": { + "function": "md5", + "args": [ + { + "attr": "tokenValue" + } + ] + } + } + } + ] + } + } +} +``` + +See [example [EX089]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/089-function-job-parameters-md5) +or an alternative [example [EX090] with SHA1 hash](https://github.com/keboola/generic-extractor/tree/master/doc/examples/090-function-job-parameters-sha1). +or an alternative [example [EX136]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/136-post-request-functions) with more deeply nested functions. + +### Optional Job Parameters +Let's say you have an API which allows you to send the list of columns to be contained in the API response. +For example, to list users and include their `id`, `name` and `login` properties, call +`/users?showColumns=id,name,login`. Also, you want to enter these values as an array in the `config` section because +the config is generated by a [template](/extend/generic-extractor/publish/). If the end-user +does not wish to filter the columns, they can +list all the columns (which would be annoying) or leave the column filter empty. In that case, the API +call would be `/users?showColumns=all`. + +The following configuration does exactly that: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/" + }, + "config": { + "columns": "", + "outputBucket": "mock-server", + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "method": "GET", + "params": { + "showColumns": { + "function": "ifempty", + "args": [ + { + "attr": "columns" + }, + "all" + ] + } + } + } + ] + } + } +} +``` + +See [example [EX097]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/097-function-ifempty). + +### User Data +Assume that you have an API returning a response that does not contain any time information. For example: + +```json +[ + { + "id": 3, + "name": "John Doe" + }, + { + "id": 234, + "name": "Jane Doe" + } +] +``` + +Add the extraction time to each record so that you at least know when each record was obtained +(when the creation time is unknown). Add additional data to each record using +the [`userData` configuration](/extend/generic-extractor/configuration/config/#user-data): + +```json +"userData": { + "extractionDate": { + "function": "date", + "args": [ + "Y-m-d H:i:s", + { + "time": "currentStart" + } + ] + } +} +``` + +The following table will be extracted: + +|id|name|extractionDate| +|3|John Doe|2017-04-20 10:17:20| +|234|Jane Doe|2017-04-20 10:17:20| + +Or, use an alternative configuration that also adds the current date: + +```json +"userData": { + "extractionDate": { + "function": "date", + "args": [ + "Y-m-d H:i:s" + ] + } +} +``` + +But whereas the first one puts a single same date to each record, the alternative configuration will return different times for different records +as they are extracted. + +See [example [EX091]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/091-function-user-data) or +an alternative [example [EX092] with a set date](https://github.com/keboola/generic-extractor/tree/master/doc/examples/092-function-user-date-set-date). + +### Headers +Suppose you have an API which requires you to send a custom `X-Api-Auth` header with every request. +The header must contain a user name and password separated by a colon. For instance, `JohnDoe:TopSecret`. + +This can be done using the following `api` configuration: + +```json +"api": { + "baseUrl": "http://example.com/", + "http": { + "headers": { + "X-Api-Auth": { + "function": "concat", + "args": [ + { + "attr": "credentials.#username" + }, + ":", + { + "attr": "credentials.#password" + } + ] + } + } + } +} +``` + +Alternatively, achieve the same result using the `implode` function: + +```json +"api": { + "baseUrl": "http://mock-server:80/093-function-api-http-headers/", + "http": { + "headers": { + "X-Api-Auth": { + "function": "implode", + "args": [ + ":", + [ + { + "attr": "credentials.#username" + }, + { + "attr": "credentials.#password" + } + ] + ] + } + } + } +} +``` + +Both configurations rely on having the username and password parameters +in the [`config` section](/extend/generic-extractor/configuration/config/), in this case also nested in the `credentials` property: + +```json +"config": { + "credentials": { + "#username": "JohnDoe", + "#password": "TopSecret" + }, + "jobs": ... +} +``` + +See [example [EX093]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/093-function-api-http-headers) or an +[alternative example [EX094] setting headers in the `config` section](https://github.com/keboola/generic-extractor/tree/master/doc/examples/094-function-config-headers). + +### Nested Functions +If the API in the [above example](#headers) tries to mimic the +[HTTP authentication](/extend/generic-extractor/configuration/api/authentication/basic/), +the header has to be sent as a [base64 encoded](https://en.wikipedia.org/wiki/Base64#MIME) value. +That is instead of sending a `JohnDoe:TopSecret`, you have to send `Sm9obkRvZTpUb3BTZWNyZXQ=`. To do this +you have to wrap the `concat` function which generates the header value in another function (`base64_encode`). + +```json +"api": { + "baseUrl": "http://example.com/", + "http": { + "headers": { + "X-Api-Auth": { + "function": "base64_encode", + "args": [ + { + "function": "concat", + "args": [ + { + "attr": "#username" + }, + ":", + { + "attr": "#password" + } + ] + } + ] + } + } + } +} +``` + +See [example [EX095]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/095-function-nested). + +### Nested StrToTime +Suppose you have an API which requires you to specify the `from` and `to` date parameters to obtain orders created +in that time interval. You want to specify only the `from` date and extract a week of data. +Enter (preferably in a [template](/extend/generic-extractor/publish/)) the +value `2017-10-04` and send an API request to +`/orders?from=2017-10-04&to=2017-10-11`. The following configuration can be used: + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/" + }, + "config": { + "startDate": "2017-10-04", + "outputBucket": "mock-server", + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "method": "GET", + "params": { + "from": { + "attr": "startDate" + }, + "to": { + "function": "date", + "args": [ + "Y-m-d", + { + "function": "strtotime", + "args": [ + "+7 days", + { + "function": "strtotime", + "args": [ + { + "attr": "startDate" + } + ] + } + ] + } + ] + } + } + } + ] + } + } +} +``` + +The configuration probably seems rather complicated, so taken apart -- the most innermost part: + +```json +{ + "function": "strtotime", + "args": [ + { + "attr": "startDate" + } + ] +} +``` + +takes the value from the `config` property `startDate` (which is `2017-10-04`) and converts it to +a timestamp value (`???` below). + +Then there is an outer part: + +```json +{ + "function": "strtotime", + "args": [ + "+7 days", + ??? + ] +} +``` + +that takes the timestamp representing `2017-10-04` and adds 7 days to it. This yields another +timestamp value (`???` below). + +Then there is another outer part: + +```json +{ + "function": "date", + "args": [ + "Y-m-d", + ??? + ] +} +``` + +converting the timestamp back to a string format (`Y-m-d` format) which yields `2017-10-11`. +This value is assigned to the `to` parameter of the API call. + +See [example [EX096]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/096-function-nested-from-to). diff --git a/src/content/docs/extend/generic-extractor/generic-intro.png b/src/content/docs/extend/generic-extractor/generic-intro.png new file mode 100644 index 000000000..c50c3e4b5 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/generic-intro.png differ diff --git a/src/content/docs/extend/generic-extractor/incremental/index.md b/src/content/docs/extend/generic-extractor/incremental/index.md new file mode 100644 index 000000000..f8d00085c --- /dev/null +++ b/src/content/docs/extend/generic-extractor/incremental/index.md @@ -0,0 +1,214 @@ +--- +title: Incremental Loading +slug: 'extend/generic-extractor/incremental' +--- + + +Extracting data incrementally is universally beneficial --- it **speeds up the extraction** and **lowers the load** on both the API and +[Keboola Storage](/storage/) (thus saving +[credits](/management/limits/#project-power)). + +## Options +After you have incrementally extracted data from an API, the data must be +[incrementally loaded](/storage/tables/#incremental-loading) +into Storage. To do that, simply set `"incrementalOutput": true` in the `config` section. + +There are, however, a number of implications in the incremental loads. It essentially boils downs to the following use cases, +depending on what kind of data you are importing (extracting from an API): + +- The imported data contains only **added entries**. When `incrementalOutput` is turned on, the data will be +simply appended to the target table in Storage. Turning `incrementalOutput` to false probably makes no sense +because the table will contain only the new entries. +- The imported data contains **added and modified entries**. When `incrementalOutput` is turned on, set a primary key on the table so that new rows are added and existing [rows are updated](/storage/tables/#primary-key-deduplication). +If the primary key is not set, the modified entries will be duplicated in the target table. Turning +`incrementalOutput` to false probably makes no sense because the table will contain only the new entries. +- The imported data contains **all rows**. In this case, set a primary key for the table or turn +`incrementalOutput` to false. Turning `incrementalOutput` to true probably makes no sense because the table will +contain duplicate entries. If you set the primary key, new rows will be added and modified rows will be updated. +Note that in this case more [credits](/management/limits/#project-power) are consumed. + +In neither of these situations will the missing rows get deleted. If you want to do so, the only way is +to turn `incrementalOutput` to false and do full loads. + +Using incremental loads obviously requires some support from the API. Generic Extractor supports incremental +loads by using [`previousStart`](/extend/generic-extractor/functions/#parameters-context) and the +[`time` function](/extend/generic-extractor/functions/#time). Setting the primary key is done using +[mappings](/extend/generic-extractor/configuration/config/mappings/). + +## Examples + +### Previous Start Example +Assume you have an API supporting a parameter `modified_since` which expects a +[Unix Timestamp](https://en.wikipedia.org/wiki/Unix_time). The response then contains only the +records that were modified after the specified date. The following configuration can be used: + +```json +{ + "config": { + "incrementalOutput": true, + "outputBucket": "mock-server", + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "modified_since": { + "time": "previousStart" + } + } + } + ] + } +} +``` + +The configuration adds the `modified_since` parameter as a reference to the internal +[`time.previousStart` value](/extend/generic-extractor/functions/#parameters-context), which contains the timestamp of the last +**successful start** of the extraction of the particular configuration. The request generated by this configuration is something like: + + GET /users?modified_since=1492606006 + +where `1492606006` is the variable timestamp of the last successful start. This introduces state into the +Generic Extractor configuration as it now remembers when it last successfully ran. This means +that if you run the above configuration every five minutes, it will extract the data modified within the last five minutes. +If you run it every hour, it will extract the data modified within the last hour. + +Should one of the runs fail or be skipped for any reason, the extraction will pick up where it ended the last time it was successful. +See [example [EX107]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/107-incremental-load). + +The last successful time is stored in the [configuration state](/extend/common-interface/config-file/#state-file). +If for some reason you need to reset it, +[update the configuration via API](https://api.keboola.com/?service=storage#put-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-). + +### Previous Start Date +If an API similar to the one in the [above example](#previous-start-example) requires the date to be +sent as a string, the following jobs configuration (which uses the [`date` function](/extend/generic-extractor/functions/#date)) +can be used: + +```json +{ + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "modified_since": { + "function": "date", + "args": [ + "Y-m-d H:i:s", + { + "time": "previousStart" + } + ] + } + } + } + ] +} +``` + +This sends a request like: + + GET /users?modified_since=2017-04-19%2012%3A46%3A46 + +in a more readable [url-decoded](https://meyerweb.com/eric/tools/dencoder/) form: + + GET /users?modified_since=2017-04-19 12:46:46 + +Otherwise the configuration behaves the same way as the [previous example](#previous-start-example). + +See [example [EX108]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/108-incremental-load-date). + +### Incremental Load From To +Another option is an API which requires the `from` and `to` parameters. The following +configuration generates the `from` date as the date of the last extraction (using the [`time.previousStart` +value](/extend/generic-extractor/functions/#parameters-context)). It also generates the `to` date as the date +of the current extraction (using the [`time.currentStart` value](/extend/generic-extractor/functions/#parameters-context)): + +```json +{ + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "from": { + "function": "date", + "args": [ + "Y-m-d", + { + "time": "previousStart" + } + ] + }, + "to": { + "function": "date", + "args": [ + "Y-m-d", + { + "time": "currentStart" + } + ] + } + } + } + ] +} +``` + +This configuration will send a request similar to this one: + + GET /109-incremental-load-from-to/users?from=2017-04-19&to=2017-04-24 + +See [example [EX109]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/109-incremental-load-from-to). + +### Incremental Relative Load +Suppose you have an API supporting the `from` and `to` parameters as in [the above example](#incremental-load-from-to) and +want to extract the last day data. It can be done using the following configuration: + +```json +{ + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "from": { + "function": "date", + "args": [ + "Y-m-d", + { + "function": "strtotime", + "args": [ + "-1 day", + { + "time": "currentStart" + } + ] + } + ] + }, + "to": { + "function": "date", + "args": [ + "Y-m-d", + { + "time": "currentStart" + } + ] + } + } + } + ] +} +``` + +This configuration leads to a request similar to this one: + + GET /110-incremental-relative/users?from=2017-04-23&to=2017-04-24 + +Remember, this is not a truly reliable incremental load. If you put such configuration +into an orchestration, and the configuration does not run for some reason, you may miss some data. +However, this may still be a useful approach for obtaining samples of data for POCs. + +See [example [EX110]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/110-incremental-relative). diff --git a/src/content/docs/extend/generic-extractor/index.md b/src/content/docs/extend/generic-extractor/index.md new file mode 100644 index 000000000..c8cd6b1b2 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/index.md @@ -0,0 +1,61 @@ +--- +title: Generic Extractor +slug: 'extend/generic-extractor' +--- + + +Generic Extractor is a [Keboola component](/overview/) that acts like a customizable +[HTTP REST](/extend/generic-extractor/tutorial/rest/) client. It can be configured to extract data +from virtually any sane web API. + +Due to the versatility of different APIs running in the wild, Generic Extractor offers many [**configuration options**](/extend/generic-extractor/configuration/). + +You may opt to use the [**visual builder**](/extend/generic-extractor/configuration/#user-interface), which provides a very convenient way +of configuring and testing the configuration. With it, you can build +an entirely new extractor for Keboola in **less than an hour**. + +![Generic Extractor - UI](/extend/generic-extractor/ui.png) + +To get started quickly, follow our [Generic Extractor tutorial](/extend/generic-extractor/tutorial). + +## Generic Extractor Requirements +Generic Extractor allows you to extract data from an API into Keboola only by configuring it. +No programming skills or additional tools are required. You just need to do two easy things before you start: + +- Become familiar with [JSON format](/extend/generic-extractor/tutorial/json/). +- Have the documentation of your chosen API at hand. The API should be [RESTful](/extend/generic-extractor/tutorial/rest/) +and, more or less, follow the HTTP specification. + +## Configuration & Development +Again, if you are new to Generic Extractor, we strongly suggest you go through the +[Generic Extractor tutorial](/extend/generic-extractor/tutorial/). It outlines the basic principles and the most important features. + +With the new convenient user interface, you can set up and test the connection in a few clicks, +just like you are used to in some other popular API development tools. + +Features such as cURL import, request tests, output mapping generator, or dynamic function templates and evaluation make the configuration process as easy as ever. + +If you intend to develop a more complicated configuration, check out how to [run Generic Extractor locally](/extend/generic-extractor/running/). +The documentation includes [several examples](https://github.com/keboola/generic-extractor/tree/master/doc) that [can also be run locally](/extend/generic-extractor/running/#running-examples). + +## Publishing Generic Extractor Configuration +Each Generic Extractor configuration can be [published](/extend/generic-extractor/publish/) as +a new standalone component. However, for registration, configurations must be +[converted to templates](/extend/generic-extractor/publish/#submission). + +Publishing your Generic Extractor configuration is **not required**. However, when published, +it can be easily used in multiple projects. A great advantage of using templates is that they +do not limit the configuration. You can always switch to JSON +[free-form configuration](/extend/generic-extractor/publish/#submission) when necessary. + +Also, templates can be used only with published components based on Generic Extractor configurations. + +## Generic Extractor Source +As with other Keboola components, the Generic Extractor connector is available on +[GitHub](https://github.com/keboola/generic-extractor/). Apart from the +main repository, it uses some vital libraries (which partially define its capabilities): + +- [Juicer](https://github.com/keboola/juicer) --- component responsible for processing HTTP JSON responses +- [CSV Map](https://github.com/keboola/php-csvmap) --- library that converts JSON data into CSV tables +- [Filter](https://github.com/keboola/php-filter) --- library that allows to match values together +- [JSON Parser](https://github.com/keboola/php-jsonparser) --- JSON parser which produces CSV tables while maintaining relations diff --git a/src/content/docs/extend/generic-extractor/map/index.md b/src/content/docs/extend/generic-extractor/map/index.md new file mode 100644 index 000000000..a1d9db550 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/map/index.md @@ -0,0 +1,235 @@ +--- +title: Generic Extractor Parameter Map +slug: 'extend/generic-extractor/map' +--- + +*To configure your first Generic Extractor, follow our [tutorial](/extend/generic-extractor/tutorial/).* + +Use the following sample configuration to navigate among various **configuration options**: + +```json +{ + "parameters": { + "api": { + "baseUrl": "https://example.com/v3.0/", + "caCertificate": "-----BEGIN CERTIFICATE-----\nMIIFaz....", + "pagination": { + "method": "multiple", + "scrollers": { + "offset_scroll": { + "method": "offset", + "offsetParam": "offset", + "limitParam": "count" + } + } + }, + "authentication": { + "type": "basic" + }, + "retryConfig": { + "maxRetries": 3 + }, + "http": { + "headers": { + "Accept": "application/json" + }, + "defaultOptions": { + "params": { + "company": 123 + } + }, + "requiredHeaders": ["X-AppKey"], + "ignoreErrors": [405], + "connectTimeout": 30, + "requestTimeout": 300 + } + }, + "aws": { + "signature": { + "credentials": { + "accessKeyId": "testAccessKey", + "#secretKey": "testSecretKey", + "serviceName": "testService", + "regionName": "testRegion" + } + } + }, + "config": { + "debug": true, + "username": "dummy", + "#password": "secret", + "outputBucket": "ge-tutorial", + "incrementalOutput": true, + "compatLevel": 2, + "http": { + "headers": { + "X-AppKey": "ThisIsSecret" + } + }, + "jobs": [ + { + "endpoint": "users", + "method": "get", + "dataField": "items", + "dataType": "users", + "params": { + "type": { + "attr": "userType" + } + }, + "responseFilter": "additional.address/details", + "responseFilterDelimiter": "/", + "scroller": "offset_scroll", + "children": [ + { + "endpoint": "users/{user_id}/orders", + "dataField": "items", + "recursionFilter": "id>20", + "placeholders": { + "user_id": "id" + } + } + ] + } + ], + "mappings": { + "content": { + "parent_id": { + "type": "user", + "mapping": { + "destination": "campaign_id", + "primaryKey": true + } + }, + "name": { + "type": "column", + "mapping": { + "destination": "text" + } + }, + "address": { + "type": "table", + "destination": "addresses", + "tableMapping": { + "street": { + "type": "column", + "mapping": { + "destination": "streetName" + } + } + } + }, + "created.date": { + "delimiter": "/", + "type": "column", + "mapping": { + "destination": "createdDate" + } + } + } + }, + "userData": { + "tag": "development" + } + }, + "iterations": [ + { + "userType": "active" + }, + { + "userType": "inactive" + } + ], + "sshProxy": { + "host": "proxy.example.com", + "user": "proxy", + "port": 22, + "#privateKey": "-----BEGIN RSA PRIVATE KEY-----\n...\n-----END RSA PRIVATE KEY-----" + } + }, + "authorization": { + "oauth_api": { + "credentials": { + "#data": "{\"status\": \"ok\",\"refresh_token\": \"1234abcd5678efgh\"}", + "appKey": "someId", + "#appSecret": "clientSecret" + } + } + } +} +``` + + + diff --git a/src/content/docs/extend/generic-extractor/publish/index.md b/src/content/docs/extend/generic-extractor/publish/index.md new file mode 100644 index 000000000..6e7df2fcb --- /dev/null +++ b/src/content/docs/extend/generic-extractor/publish/index.md @@ -0,0 +1,347 @@ +--- +title: Publish Generic Extractor +slug: 'extend/generic-extractor/publish' +redirect_from: + - /extend/generic-extractor/registration/ +--- + + +It is possible to publish a Generic Extractor configuration as a completely separate component. +This enables sharing the API extractor between various projects and simplifies its further configuration. + +## Configuration Considerations +Before converting your configuration to a universally available component, consider +what values in the configuration should be provided by the end-user (typically authentication values). +Then design a [configuration schema](/extend/component/ui-options/configuration-schema/) for setting +those values. You can [test the schema online](http://jeremydorn.com/json-editor/) ([alternative](https://mozilla-services.github.io/react-jsonschema-form/)). +The values obtained from the end user will be stored in the [`config` property](/extend/generic-extractor/configuration/config/). +Modify your configuration to read those values from there. + +Do not forget that if you prefix a value with a hash `#`, it will be +[encrypted](/overview/encryption/) once the configuration is saved. +Also, try to make the extractor [work incrementally](/extend/generic-extractor/incremental/) +if possible. + +## Publishing +To publish your Generic Extractor configuration, you need to [create a new component](/extend/component/tutorial/) in +the [Developer Portal](https://components.keboola.com/). Choose an appropriate name and the type `extractor`. Once you +have created the component, edit it, and fill in the following details: + +- **Repository** + - **Type** --- AWS ECR + - **Image Name** -- `147946154733.dkr.ecr.us-east-1.amazonaws.com/developer-portal-v2/ex-generic-v2` + - **Tag** -- see the [Generic Extractor GitHub repository](https://github.com/keboola/generic-extractor/releases) + - **Region** -- leave empty +- **UI options** --- set to `genericTemplatesUI` + +For a list of available tags, see the [Generic Extractor GitHub repository](https://github.com/keboola/generic-extractor/). It is also possible to use the `latest` tag, which points to the highest available tag. However, +we recommend that you configure your component with a specific tag and update it manually to avoid problems with breaking changes +in future Generic Extractor releases. + +Because the UI is assumed to be `genericTemplatesUI`, provide a +[**configuration schema**](/extend/component/ui-options/configuration-schema/) and +a **template** to be used in conjunction with the schema. Optionally, the template UI may also contain an interface to +negotiate [OAuth authentication](/extend/generic-extractor/configuration/api/authentication/#oauth). +An example of the template UI is shown in the picture below. + +![Screenshot - Generic templates UI](/extend/generic-extractor/template-1.png) + +The `Config` section of the templates UI is defined by the configuration schema you provide. +The `Template` section contains at least one template. A template is simply a configuration of +Generic Extractor. + +For example, you might want to provide one configuration for incremental loading +and a different configuration for full loading. The template UI also has the option to +`Switch to JSON editor`, which displays the configuration JSON and allows the end user to modify it. +Notice that the JSON editor allows modification only to the [`config`](/extend/generic-extractor/configuration/config/) +section. Other sections, such as [`api`](/extend/generic-extractor/configuration/api/) or +[`authorization.oauth_api`](/extend/generic-extractor/configuration/api/authentication/#oauth), may not be modified by the end user. + +You can review existing templates in their [GitHub repository](https://github.com/keboola/kbc-ui-templates/tree/master/resources). +If you feel confident, you can send a pull request with your templates, otherwise submit it when requesting the +[publication of your component](/extend/publish/). + +## Example +Let's say you have the following working API configuration +(see [example [EX111]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/111-templates-example)): + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/", + "authentication": { + "type": "login", + "loginRequest": { + "endpoint": "token", + "headers": { + "Authorization": { + "function": "base64_encode", + "args": [ + "JohnDoe:TopSecret" + ] + } + } + }, + "apiRequest": { + "headers": { + "X-Api-Auth": "auth.token" + } + } + }, + "default": { + "http": { + "params": { + "accountId": 123 + } + } + } + }, + "config": { + "incrementalOutput": true, + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "type": "active" + } + }, + { + "endpoint": "orders", + "dataType": "orders" + } + ] + } + } +} +``` + +and you identify that four values of that configuration need to be specified by the end user: +`JohnDoe`, `TopSecret`, `123`, and `active`. + +For each of the values, create a parameter of the appropriate type: + +- `JohnDoe` --- a string parameter `login` +- `TopSecret` --- a string parameter `#password` (it will be encrypted) +- `123` --- a numeric parameter `accountId` +- `active` --- an enumeration parameter `userType` with values `active`, `inactive`, `all` + +The parameter names are completely arbitrary. However, they must not conflict with existing +configuration properties of [Generic Extractor](/extend/generic-extractor/configuration/config/) (e.g., `jobs`, `mappings`). +Now create a [configuration schema](/extend/component/ui-options/configuration-schema/) for the four parameters. + +```json +{ + "title": "Person", + "type": "object", + "properties": { + "login": { + "type": "string", + "title": "Login:", + "description": "Your API user name", + "minLength": 4 + }, + "#password": { + "type": "string", + "title": "Password:", + "description": "Your API password", + "minLength": 4 + }, + "accountId": { + "type": "integer", + "title": "Account ID", + "description": "See in-app help for obtaining Account Id" + }, + "userType": { + "title": "User type:", + "type": "string", + "enum": [ + "active", + "inactive", + "all" + ], + "default": "active", + "description": "Specify which users to obtain" + } + }, + "required": [ + "login", "#password", "accountId", "userType" + ] +} +``` + +When you test the [schema online](http://jeremydorn.com/json-editor/) ([alternative](https://mozilla-services.github.io/react-jsonschema-form/)), it will produce a +configuration JSON: + +![Screenshot - Schema Test](/extend/generic-extractor/schema-test.png) + +```json +{ + "login": "JohnDoe", + "#password": "TopSecret", + "accountId": 123, + "userType": "inactive" +} +``` + +The above properties will be merged into the [`config` section](/extend/generic-extractor/configuration/config/). Now +modify the configuration so that it reads them from there using [functions and references](/extend/generic-extractor/functions/). + +```json +{ + "parameters": { + "api": { + "baseUrl": "http://example.com/", + "authentication": { + "type": "login", + "loginRequest": { + "endpoint": "token", + "headers": { + "Authorization": { + "function": "base64_encode", + "args": [ + { + "function": "concat", + "args": [ + { + "attr": "username" + }, + ":", + { + "attr": "#password" + } + ] + } + ] + } + } + }, + "apiRequest": { + "headers": { + "X-Api-Auth": "auth.token" + } + } + } + }, + "config": { + "incrementalOutput": true, + "username": "JohnDoe", + "#password": "TopSecret", + "accountId": 123, + "userType": "active", + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "accountId": { + "attr": "accountId" + }, + "type": { + "attr": "userType" + } + } + }, + { + "endpoint": "orders", + "dataType": "orders", + "params": { + "accountId": { + "attr": "accountId" + } + } + } + ] + } + } +} +``` + +The argument to the `base64_encode` function is now the +[`concat` function](/extend/generic-extractor/functions/#concat), which joins together the +values of the `username` and `#password` fields. The `accountId` parameter needs to be moved to the +`jobs` section because the `http.defaultOptions.params` section does not support function calls (yet!). +The `type` parameter was changed to a reference to the `userType` field +(see [example [EX111]](https://github.com/keboola/generic-extractor/tree/master/doc/examples/111-templates-example)). + +When you handled the configuration parameters, turn the configuration into a template. Place +the `api` section to a separate, individual `api.json` file: + +```json +{ + "baseUrl": "http://example.com/", + "authentication": { + "type": "login", + "loginRequest": { + "endpoint": "token", + "headers": { + "Authorization": { + "function": "base64_encode", + "args": [ + { + "function": "concat", + "args": [ + { + "attr": "username" + }, + ":", + { + "attr": "#password" + } + ] + } + ] + } + } + }, + "apiRequest": { + "headers": { + "X-Api-Auth": "auth.token" + } + } + } +} +``` + +Once you make sure that the extractor works as it did before, +remove the user provided values (`username`, `#password`, `accountId`, `userType`) from +the `config` section, put it in a `data` section and add `name` and `description` to it. +Save the file into a separate `template.json` file. The template file therefore contains +`name`, `description` and `data` nodes. + +```json +{ + "name": "Basic", + "description": "Basic incremental template", + "data": { + "incrementalOutput": true, + "jobs": [ + { + "endpoint": "users", + "dataType": "users", + "params": { + "accountId": { + "attr": "accountId" + }, + "type": { + "attr": "userType" + } + } + }, + { + "endpoint": "orders", + "dataType": "orders", + "params": { + "accountId": { + "attr": "accountId" + } + } + } + ] + } +} +``` + +Create as many `template.json` files as you wish. However, all of them need to share the same `api.json` +configuration. When you want to publish your component, attach the `api.json` and all `template.json` files. diff --git a/src/content/docs/extend/generic-extractor/running/index.md b/src/content/docs/extend/generic-extractor/running/index.md new file mode 100644 index 000000000..4a5c40ab6 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/running/index.md @@ -0,0 +1,155 @@ +--- +title: Running Generic Extractor +slug: 'extend/generic-extractor/running' +--- + + +Generic Extractor is normally run from within the Keboola user interface. It can be found in the **Extractors** section +and all you need to do is provide its configuration JSON. No other settings are necessary. + +![Screenshot - Generic Extractor Configuration](/extend/generic-extractor/configuration.png) + +Because creating the configuration JSON can be a non-trivial task, there are some things which can help +you in developing the configuration. + +## Debug Mode +Debug mode can be turned on by setting `"debug": true` in the `config` section of the configuration, e.g.: + +```json +{ + "api": { + ... + }, + "config": { + "debug": true, + ... + } +} +``` + +In debug mode, the extractor displays all API requests it sends, helping you understand what is really happening, +why something is skipped, etc. + +![Screenshot - Debug Logs](/extend/generic-extractor/events.png) + +**Warning:** If the API sends sensitive data (e.g. authorization token) in the URL, these may become +visible in the events. Also, debug mode considerably slows the extraction. Therefore it should never +be turned on in production configurations. + +## Running Locally +If you are working on a complicated configuration, or developing a new component based on +Generic Extractor, running every configuration from the Keboola UI may be slow and tedious. +You may run Generic Extractor locally, provided that you have access to Docker. +The following is **not necessary** to run or configure Generic Extractor in Keboola. + +### Run Built Version +Create an empty directory somewhere and in it create a `config.json` file with a +configuration you want to execute. For example: + +```json +{ + "parameters": { + "api": { + "baseUrl": "https://api.github.com", + "http": { + "Accept": "application/json", + "Content-Type": "application/json;charset=UTF-8" + } + }, + "config": { + "debug": true, + "jobs": [ + { + "endpoint": "/orgs/keboola/members", + "dataType": "members" + } + ] + } + } +} +``` + +Then run Generic Extractor in the current directory by executing the following command on *nix systems: + + docker run -v $(pwd):/data 147946154733.dkr.ecr.us-east-1.amazonaws.com/developer-portal-v2/ex-generic-v2:latest + +or on Windows: + + docker run -v %cd%:/data 147946154733.dkr.ecr.us-east-1.amazonaws.com/developer-portal-v2/ex-generic-v2:latest + +You should see: + + DEBUG: Using NO Auth [] [] + DEBUG: Using automatic conversion of single values to arrays where required. [] [] + DEBUG: GET /orgs/keboola/members HTTP/1.1 Host: api.github.com User-Agent: Guzzle/5.3.1 curl/7.38.0 PHP/7.0.17 [] [] + DEBUG: Analyzing members {"rowsAnalyzed":[],"rowsToAnalyze":7} [] + DEBUG: Processing results for __kbc_default. [] [] + INFO: Extractor finished successfully. [] [] + +along with the output tables created in `/out/tables` sub-directory of the current directory. +It is recommended to remove the contents of the `out/tables` directory before running the extractor again. + +**Important:** Generic Extractor itself is not able to decrypt encrypted values. That means that when you +supply the configuration directly in the `config.json` file, you must always provide decrypted values --- e.g.: + +```json +{ + ..., + "config": { + "#username": "JohnDoe", + "#password": "TopSecret", + ... + } +} +``` + +When you store such configuration in the Keboola UI, it will automatically be encrypted: + +```json +{ + ..., + "config": { + "#username": "JohnDoe", + "#password": "KBC::ComponentProjectEncrypted==r13Khq0lR4ycDNTujirz5/GMqNEVZ4tZ2OTmRcsNYqlP/a/STMelWtz9R8yEtr3ck6KiYA7XrL8pqIQv9S7Ro28KNZgmqtSNzKhFcEsItPnTDCQqvnU99q2a0ES+oN/v", + ... + } +} +``` + +The above configuration then **cannot** be run locally. +Read more about [encryption](/overview/encryption/). + +### Building and Running the Image +To build the container from source: + +- Clone this repository: `git clone https://github.com/keboola/generic-extractor.git`. +- Switch to the created directory: `cd generic-extractor`. +- Build the container: `docker compose build`. +- Install dependencies locally: `docker compose run --rm extractor composer install`. +- Create a **data folder** for configuration: `mkdir data`. + +To run the built container: + +- Create a configuration file `config.json` in the **data folder**. +- Run the extraction: `docker compose run --rm extractor`. +- You will find the extracted data in the `out/tables` sub-directory of the **data folder**. + +Before running the extractor again, it is recommended to clear the `out` directory by +running `docker compose run --rm extractor rm -rf data/out`. + +## Running Examples +[All examples](https://github.com/keboola/generic-extractor/tree/master/doc) referenced in this documentation are actually runnable against the proper API. Because +it is difficult to find the specific API for the case (and gain access to it), you can test +these configurations against a [mock server](https://github.com/keboola/ex-generic-mock-server). +Each example contains a set of requests (`*.request` file) and responses (`*.response`) and +optionally their headers (`*.requestHeaders` and `*.responseHeaders`). + +To run the examples: + +- Clone Generic Extractor repository: `git clone https://github.com/keboola/generic-extractor.git`. +- Navigate to the documentation directory: `cd generic-extractor/doc`. +- Run a single example of your choice, e.g.: `docker compose run -e "KBC_EXAMPLE_NAME=001-simple-job" extractor`. +- The output will be available in `examples/001-simple-job/out/tables`. +- Or run all examples by executing `./run-samples.sh`. + +If you want to create your own example, follow the instructions in the [mock server repository](https://github.com/keboola/ex-generic-mock-server/blob/master/README.md#creating-examples). diff --git a/src/content/docs/extend/generic-extractor/schema-test.png b/src/content/docs/extend/generic-extractor/schema-test.png new file mode 100644 index 000000000..f141f29f7 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/schema-test.png differ diff --git a/src/content/docs/extend/generic-extractor/template-1.png b/src/content/docs/extend/generic-extractor/template-1.png new file mode 100644 index 000000000..bce5ad800 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/template-1.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/2_child.png b/src/content/docs/extend/generic-extractor/tutorial/2_child.png new file mode 100644 index 000000000..2678c3f85 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/2_child.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/base_configuration.png b/src/content/docs/extend/generic-extractor/tutorial/base_configuration.png new file mode 100644 index 000000000..2f5ca6297 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/base_configuration.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/basic/index.md b/src/content/docs/extend/generic-extractor/tutorial/basic/index.md new file mode 100644 index 000000000..0d5b34e09 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/tutorial/basic/index.md @@ -0,0 +1,216 @@ +--- +title: Basic Configuration +slug: 'extend/generic-extractor/tutorial/basic' +--- + + +Before configuring Generic Extractor, you should have a basic understanding +of [REST API](/extend/generic-extractor/tutorial/rest/) and +[JSON format](/extend/generic-extractor/tutorial/json/). This tutorial uses the +[MailChimp API](https://mailchimp.com/developer/reference/), so +have its documentation at hand. You also need the +[MailChimp API key](/extend/generic-extractor/tutorial/#prepare). + +## Configuration +Generic Extractor configuration is written in [JSON format](/extend/generic-extractor/tutorial/json/) +and comprises [several sections](/extend/generic-extractor/configuration/#configuration-sections) (a +[configuration map](/extend/generic-extractor/map/) for navigation is available). + +A [user interface](/extend/generic-extractor/configuration/#user-interface) is available that can help you with the configuration + and generate the JSON configuration for you. + +### Base Configuration +The first configuration part is a `Base Configuration` section where you can set the Base URL and Authentication method of the +API you connect to. + +In our case, we will use the MailChimp API, so the `Base URL` will be `https://us13.api.mailchimp.com/3.0/`, and the `Authentication` method will be `Basic Authentication`. + +**Important:** Make sure that the `baseUrl` URL ends with a slash! + +In the `Destination` section, you can set: +- The `Output Bucket` where the data will be stored. It will be set to the ID of the [Storage Bucket](/storage/buckets/) +- `Incremental Output` option, which defines whether you want the result to overwrite the existing data or append to it. [See more](/extend/generic-extractor/incremental/) + - Note that when using Incremental Output, you should set up the mapping. + +![Base Configuration](/extend/generic-extractor/tutorial/base_configuration.png) + +#### JSON +If you switch to the `JSON` mode, the created configuration will translate to the `api` section where you set the **basic properties** of the API. +In the most simple case, this is the `baseUrl` property and `authentication`, as shown in this JSON snippet: + +```json +{ + "api": { + "baseUrl": "https://us13.api.mailchimp.com/3.0/", + "authentication": { + "type": "basic" + } + } +} +``` + +**Important:** Make sure that the `baseUrl` URL ends with a slash! + +The `config` section describes the **actual extraction**. Its most important parts are the `outputBucket` and +`jobs` properties. `outputBucket` must be set to the ID of the [Storage Bucket](/storage/buckets/) +where the data will be stored. If no bucket exists, it will be created. + +It also contains the authentication parameters, such as `username` and `password`. Start with this +configuration section: + +```json +"config": { + "username": "dummy", + "#password": "c40xxxxxxxxxxxxxxxxxxxxxxxxxxxxx-us13", + "outputBucket": "ge-tutorial", + "incrementalOutput": false +} +``` + +The `password` property is prefixed with the hash mark `#`, meaning the value will be [encrypted](/overview/encryption/) once +you save the configuration. + +### Endpoint Section +Once you set up the Base Configuration, you can set up the actual endpoint to be queried. + +Start by clicking the **+ New Endpoint** button: + +![New Endpoint](/extend/generic-extractor/tutorial/new_endpoint.png) + +You will be asked to provide the relative endpoint URL path. In our case, we will use the `campaigns` endpoint. + +![New Endpoint modal](/extend/generic-extractor/tutorial/new_endpoint_modal.png) + +- In the URL section, you will see the resulting endpoint URL combined with the `Base URL` you set up in the `Base Configuration` section. + - **Important:** Do not start the URL with a slash. If you do so, the URL +will be absolute from the domain `https://us13.api.mailchimp.com/campaigns`, which is invalid +(it is missing the `3.0` part). An alternative would be to put `/3.0/campaigns` in the `endpoint` property. +- Alternatively, you may opt to create the endpoint using the **cURL command**, which is usually available in the API documentation. + +Now you are getting close to a runnable configuration, and you may proceed with testing the configuration by clicking the `TEST ENDPOINT` button: + +![Test endpoint](/extend/generic-extractor/tutorial/test_endpoint.png) + +In the test endpoint popup, you will see the following sections: +- `Records`: The actual data that will be used for parsing. +- `Response`: The response from the API. It includes headers, status code, and response body in the `data` property. +- `Request`: The request that has been sent to the API. +- `Debug log`: A log outputted by the component for debugging purposes. + +In the `Records` section, you will now see the following: +``` +[ + "The root element of the response is not a list; please change your Data Selector path to list" +] +``` + +Also, if you try to run this configuration, you will get an error similar to this: + + The response contains more than one array! Use the 'dataField' parameter to specify a key to the data array. + (endpoint: campaigns, arrays in the response root: campaigns, _links) + +That means that the extractor got the response but cannot automatically process it. The `Data Selector` path doesn't point to an array. + +Examine the `data` attribute of the response, and you will see the following objects: `campaigns`, `total_items`, and `_links`: + +```json +{ + "campaigns": [ + { + "id": "42694e9e57", + "type": "regular", + ... + }, + { + "id": "f6276207cc", + "type": "regular", + ... + } + ], + "total_items": 2, + "_links": [ + { + "rel": "parent", + "href": "https://usX.api.mailchimp.com/3.0/", + "method": "GET", + "targetSchema": "https://api.mailchimp.com/schema/3.0/Root.json" + }, + { + "rel": "self", + "href": "https://usX.api.mailchimp.com/3.0/campaigns", + "method": "GET", + "targetSchema": "https://api.mailchimp.com/schema/3.0/Campaigns/Collection.json", + "schema": "https://api.mailchimp.com/schema/3.0/CollectionLinks/Campaigns.json" + } + ] +} +``` + +Generic Extractor expects the response to be an array of items. If it receives an object, it +searches its properties to find an array. Finding multiple arrays will be confusing because it is unclear which array you want. +To fix this, change the `Data Selector` parameter (aka `dataField`) to value `campaigns` to point to the array of items you want to extract. + +![Selector](/extend/generic-extractor/tutorial/data_selector.png) + +Now, run the configuration by clicking the **Run** button and go to the job details to see what happened: + +![Screenshot - Generic Extractor job](/extend/generic-extractor/tutorial/job-1.png) + +The extraction produced two tables. The `in.c-ge-tutorial.campaigns` table contains all the +fields of a campaign and as many rows as you have campaigns. + +![Screenshot - Campaigns Table](/extend/generic-extractor/tutorial/table-campaigns-sample.png) + +The table `in.c-ge-tutorial.campaigns__links` contains the contents of the `_links` property. +Because the `_links` property is a nested array within a single campaign object, it cannot be easily +represented in a single column of the `campaigns` table. Generic Extractor, therefore, replaces the column +value with a generated key, for example, `campaigns_75d5b14d79d034cd07a9d95d5f0ca5bd`, and automatically +creates a new table that has the column `JSON_parentId` with that value so that you can join the tables together. + +### Final JSON Configuration +The main parts of the configuration and their nesting are shown in the following schema: + +![Schema - Generic Extractor configuration](/extend/generic-extractor/generic-intro.png) + +The resulting JSON configuration will look like this: + +```json +{ + "parameters": { + "api": { + "baseUrl": "https://us13.api.mailchimp.com/3.0/", + "authentication": { + "type": "basic" + } + } + "config": { + "username": "dummy", + "#password": "c40xxxxxxxxxxxxxxxxxxxxxxxxxxxxx-us13", + "outputBucket": "ge-tutorial", + "jobs": [ + { + "endpoint": "campaigns", + "dataField": { + "path": "campaigns", + "delimiter": "." + } + } + ] + } + } +} +``` + +**Important:** It may seem confusing that the `endpoint` and `dataField` properties are set to `campaigns`. +This is just a coincidence; the `endpoint` property refers to the `campaigns` in the resource URL, and +the `dataField` refers to the `campaigns` property in the JSON retrieved as the API response. + +## Summary +The above tutorial demonstrates a very basic configuration of Generic Extractor. The extractor is capable +of doing much more; see other parts of this tutorial for an explanation of pagination, jobs and mapping: + +- [Pagination](/extend/generic-extractor/tutorial/pagination/) --- breaks a result with many items into separate pages. +- [Jobs](/extend/generic-extractor/tutorial/jobs/) --- describe the API endpoints + (resources) to be extracted. +- [Mapping](/extend/generic-extractor/tutorial/mapping/) --- describes how the JSON + response is converted into CSV files that will be imported into Storage. diff --git a/src/content/docs/extend/generic-extractor/tutorial/child_debug.png b/src/content/docs/extend/generic-extractor/tutorial/child_debug.png new file mode 100644 index 000000000..926488ff5 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/child_debug.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/child_endpoint.png b/src/content/docs/extend/generic-extractor/tutorial/child_endpoint.png new file mode 100644 index 000000000..b431a9b81 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/child_endpoint.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/config-1.png b/src/content/docs/extend/generic-extractor/tutorial/config-1.png new file mode 100644 index 000000000..55f407d96 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/config-1.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/configuration-schema.svg b/src/content/docs/extend/generic-extractor/tutorial/configuration-schema.svg new file mode 100644 index 000000000..0c010c242 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/tutorial/configuration-schema.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/src/content/docs/extend/generic-extractor/tutorial/create_endpoint_child.png b/src/content/docs/extend/generic-extractor/tutorial/create_endpoint_child.png new file mode 100644 index 000000000..07dbf2f49 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/create_endpoint_child.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/create_mapping.png b/src/content/docs/extend/generic-extractor/tutorial/create_mapping.png new file mode 100644 index 000000000..8fa62ba38 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/create_mapping.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/create_mapping_toggle.png b/src/content/docs/extend/generic-extractor/tutorial/create_mapping_toggle.png new file mode 100644 index 000000000..bf1c4d65f Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/create_mapping_toggle.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/data_selector.png b/src/content/docs/extend/generic-extractor/tutorial/data_selector.png new file mode 100644 index 000000000..91083e09b Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/data_selector.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/img.png b/src/content/docs/extend/generic-extractor/tutorial/img.png new file mode 100644 index 000000000..02755a43f Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/img.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/img_1.png b/src/content/docs/extend/generic-extractor/tutorial/img_1.png new file mode 100644 index 000000000..70e32bd77 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/img_1.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/index.md b/src/content/docs/extend/generic-extractor/tutorial/index.md new file mode 100644 index 000000000..181070f9a --- /dev/null +++ b/src/content/docs/extend/generic-extractor/tutorial/index.md @@ -0,0 +1,80 @@ +--- +title: Generic Extractor Tutorial +slug: 'extend/generic-extractor/tutorial' +--- + + +In this tutorial, we will guide you through configuring Generic Extractor for a new API. +In our case, MailChimp --- an email marketing service. + +Even though there already is a MailChimp extractor available in Keboola as a +[component](/extend/generic-extractor/publish/) based on Generic Extractor, +the [MailChimp](https://mailchimp.com/) API is ideal for this tutorial because it is fairly +easy to understand and has excellent documentation. + +## Prepare +There are a few things you need to do before you start: + +1. Read our [quick introduction to REST](/extend/generic-extractor/tutorial/rest/) for a basic understanding of +**HTTP requests** and **REST API**. +2. Read our [quick introduction to JSON](/extend/generic-extractor/tutorial/json/) to learn how to write **JSON +configurations**. +3. [Create your MailChimp account](https://login.mailchimp.com/signup/), free of charge, if you do not have one +already. +4. Follow the MailChimp wizard or [help](https://us13.admin.mailchimp.com/campaigns/) and **fill the account with +data**: + - Create a new Campaign (choose the *regular type*). + - Create a new List and add some addresses to it (preferably yours). + - Go back to Campaigns, select your campaign and hit "Next" in the bottom right corner. + - Design a test email and send it. + - Check that you have received the email and read it. +5. To gain access to the MailChimp API, go to your Account detail and under Extras find the option to +[generate your API Key](https://mailchimp.com/help/about-api-keys/#Find-or-Generate-Your-API-Key). +It will look like this: `c40xxxxxxxxxxxxxxxxxxxxxxxxxxxxx-us13`. + +## Get Started +Let's take a closer look at the [MailChimp API](https://mailchimp.com/developer/) now. +There are plenty of documentation guides available. To explore the API and review what information is in +each resource, use, for example, the [Playground](https://us1.api.mailchimp.com/playground/). + +The basic properties of the API are outlined in the +[Getting Started Guide](https://mailchimp.com/developer/guides/get-started-with-mailchimp-api-3/#resources). +The following are the crucial parts for our use-case: + +- The root API URL is `https://.api.mailchimp.com/3.0`, where `` refers to a data center for your +account. The data center is the last part of the API key; if the API key is +`c40xxxxxxxxxxxxxxxxxxxxxxxxxxxxx-us13`, the root URL is `https://us13.api.mailchimp.com/3.0`. +- API Authentication can be done using **HTTP Basic Authentication** where you use **any string** (text) for +username and the API key for password. + +Now, go straight to the documentation of the +[**Campaign** resource](https://mailchimp.com/developer/reference/campaigns/). +Because you intend to extract data from MailChimp, the only part you are interested in is the **Read Method**. + +![Screenshot - Read Campaign Documentation](/extend/generic-extractor/tutorial/mailchimp-api-docs-1.png) + +The documentation lists the URL (`/campaigns`) of the **Campaign Resource**, and the query string +parameters (these go into the URL), such as `fields`, `count`, etc. It also lists example +requests and responses. The response body is in [JSON](/extend/generic-extractor/tutorial/json) format and starts like this: + +```json +{ + "campaigns": [ + { + "id": "42694e9e57", + "type": "regular", + "create_time": "2015-09-15T14:40:36+00:00", + ... +``` + +## Next Steps +Now you have everything you need to actually start extracting the data. Continue with your Generic Extractor +configuration here: + +- [Basic configuration](/extend/generic-extractor/tutorial/basic/) --- sets the basic properties of the API and describes the actual extraction. +- [Pagination](/extend/generic-extractor/tutorial/pagination/) --- breaks a result with a + large number of items into separate pages. +- [Jobs](/extend/generic-extractor/tutorial/jobs/) --- describes the API endpoints + (resources) to be extracted. +- [Mapping](/extend/generic-extractor/tutorial/mapping/) --- describes how the JSON + response is converted into CSV files that will be imported into Storage. \ No newline at end of file diff --git a/src/content/docs/extend/generic-extractor/tutorial/job-1.png b/src/content/docs/extend/generic-extractor/tutorial/job-1.png new file mode 100644 index 000000000..4b784849f Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/job-1.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/job-2.png b/src/content/docs/extend/generic-extractor/tutorial/job-2.png new file mode 100644 index 000000000..1b113919a Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/job-2.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/job-table-1.png b/src/content/docs/extend/generic-extractor/tutorial/job-table-1.png new file mode 100644 index 000000000..fedfc0f45 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/job-table-1.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/job-table-2.png b/src/content/docs/extend/generic-extractor/tutorial/job-table-2.png new file mode 100644 index 000000000..49237514c Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/job-table-2.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/jobs/index.md b/src/content/docs/extend/generic-extractor/tutorial/jobs/index.md new file mode 100644 index 000000000..ad386df77 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/tutorial/jobs/index.md @@ -0,0 +1,238 @@ +--- +title: Jobs Tutorial +slug: 'extend/generic-extractor/tutorial/jobs' +--- + + +On your way through the Generic Extractor tutorial, you have learned about + +- [Basic configuration](/extend/generic-extractor/tutorial/basic/) and +- [Configuration of pagination](/extend/generic-extractor/tutorial/pagination/). + +Now, we will show you how to use Generic Extractor's **sub-jobs**. + +Let's start this section with a closer examination of the `campaigns` resource of the MailChimp API. +Besides retrieving multiple campaigns using the `/campaigns` endpoint, it can also retrieve detailed +information about a single campaign using `/campaigns/{campaign_id}`. + +![Screenshot - Mailchimp documentation](/extend/generic-extractor/tutorial/mailchimp-api-docs-1.png) + +Moreover, each campaign has three **sub-resources**: +`/campaigns/{campaign_id}/content`, `/campaigns/{campaign_id}/feedback` +and `/campaigns/{campaign_id}/send-checklist`. The `{campaign_id}` expression represents a placeholder +that a specific campaign ID should replace. To retrieve the sub-resource, use child jobs. + +## Child Jobs + +In the +[previous part](/extend/generic-extractor/tutorial/pagination/#running) of the tutorial, you created this job +property in the Generic Extractor configuration: + +```json +"jobs": [ + { + "endpoint": "campaigns", + "dataField": "campaigns" + } +] +``` + +All sub-resources are retrieved by configuring the `children` property in JSON; its structure is the same as the +structure of the `jobs` property, but it must additionally define `placeholders`. + +In the UI, you just create a new endpoint and mark it as a `Child Job` of the parent job of your choice. Any placeholders, +e.g., variables that will be set from the parent object, should be enclosed in curly braces, e.g., `{campaign_id}`. + +![Create endpoint](/extend/generic-extractor/tutorial/create_endpoint_child.png) + +Once the endpoint is created, the `Placeholders section` will be prefilled for you. We will set the `Response Path` value to `id`, +since we want to use the `id` property from the parent response to replace the `{campaign_id}` placeholder in the child endpoint. + +![Child endpoint](/extend/generic-extractor/tutorial/child_endpoint.png) + +Now, you can test the endpoint as in previous examples. +The `Mapping.Data Selector` (aka `dataField`) property must refer to an array, i.e., `items` or `_links` in our case +(see the [documentation](https://mailchimp.com/developer/reference/campaigns/campaign-checklist/)). + +When you look at the debug log, you will also see that the connector is making all the parent requests: + +![child_debug](/extend/generic-extractor/tutorial/child_debug.png) + +**The resulting underlying JSON will look like this:** + +```json +"jobs": [ + { + "endpoint": "campaigns", + "dataField": "campaigns", + "children": [ + { + "endpoint": "campaigns/{campaign_id}/send-checklist", + "dataField": "items", + "placeholders": { + "campaign_id": "id" + } + } + ] + } +] +``` + +The `children` are executed for each element retrieved from the parent endpoint, i.e., for each campaign. +The `placeholders` setting connects the placeholders used in the `endpoint` property with +the data in the actual parent response. +That means that the `campaign_id` placeholder in the `campaigns/{campaign_id}/send-checklist` endpoint +will be replaced by the `id` property of the JSON [response](https://mailchimp.com/developer/reference/campaigns/): + +![Screenshot - Mailchimp docs](/extend/generic-extractor/tutorial/mailchimp-api-docs-2.png) + +Also, note that the placeholder name is completely arbitrary (i.e., it is just a coincidence that +it is also named `campaign_id` in the Mailchimp documentation). Therefore, the following configuration is +also valid: + +```json +{ + "parameters": { + "api": { + "baseUrl": "https://us13.api.mailchimp.com/3.0/", + "authentication": { + "type": "basic" + }, + "pagination": { + "method": "offset", + "offsetParam": "offset", + "limitParam": "count", + "limit": 1 + } + }, + "config": { + "debug": true, + "username": "dummy", + "#password": "c40xxxxxxxxxxxxxxxxxxxxxxxxxxxxx-us13", + "outputBucket": "ge-tutorial", + "jobs": [ + { + "endpoint": "campaigns", + "dataField": "campaigns", + "children": [ + { + "endpoint": "campaigns/{cid}/send-checklist", + "dataField": "items", + "placeholders": { + "cid": "id" + } + } + ] + } + ] + } + } +} +``` + +Running the above configuration gives you a new table named, for example, +`in.c-ge-tutorial.campaigns__campaign_id__send-checklist`. The table +contains messages from campaign checking. You will see something like this: + +![Screenshot - Job Table](/extend/generic-extractor/tutorial/job-table-1.png) + +Note that apart from the API response properties `type`, `heading`, and `details`, an additional field, +`parent_id`, was added. It contains the value of the placeholder (`campaign_id`) for the particular +request. So, to join the two tables together in SQL, you would use the join condition: + + campaigns.id=campaigns__campaign_id__send-checklist.parent_id + +However, you have to remember what table the `parent_id` column refers to. + +## Multiple Jobs +You have probably noticed that the `jobs` and `children` properties are arrays. It means that you can retrieve multiple +endpoints in a single configuration. Let's pick the campaign `content` sub-resource too: + +![second child](/extend/generic-extractor/tutorial/2_child.png) + +The placeholder configuration is the same, however, +the question is what to put in the `Data Selector` (`dataField`). If you examine the sample [response](https://mailchimp.com/developer/reference/campaigns/campaign-content/) +after running the test endpoint, it looks like this: + +```json + +{ + "plain_text": "** Designing...*|END:IF|*", + "html": "", + "_links": [ + { + "rel": "parent", + "href": "https://usX.api.mailchimp.com/3.0/campaigns/42694e9e57", + "method": "GET", + "targetSchema": "https://api.mailchimp.com/schema/3.0/Campaigns/Instance.json" + }, + ... + ] +} + +``` + +If you use JSON configuration with no `dataField` like in the above configuration and run it, you will obtain a table like this: + +![Screenshot - Job Table](/extend/generic-extractor/tutorial/job-table-2.png) + +This is not what you expected. Instead of obtaining the campaign content, you +got the `_links` property from the response because Generic Extractor automatically +picks an array in the response. To get the entire response as a **single table record**, set `dataField` +to the [path](/extend/generic-extractor/tutorial/json/#references) in the object. Because you want to use the +**entire response**, set `dataField` to `.` to start in the root. + +***Note:** If you use the UI editor, the `Data Selector` (`dataField`) is automatically set to `.` by default.* + +**The resulting JSON:** + +```json +"jobs": [ + { + "endpoint": "campaigns", + "dataField": "campaigns", + "children": [ + { + "endpoint": "campaigns/{campaign_id}/send-checklist", + "dataField": { + "path": "items", + "delimiter": "." + }, + "placeholders": { + "campaign_id": "id" + } + }, + { + "endpoint": "campaigns/{campaign_id}/content", + "dataField": { + "path": ".", + "delimiter": "." + }, + "placeholders": { + "campaign_id": "id" + } + } + ] + } +] +``` + +Running the above configuration will get you the table `in.c-ge-tutorial.campaigns__campaign_id__content` +with columns like `plain_text`, `html`, and others. + +You will also get the table `in.c-ge-tutorial.campaigns__campaign_id__content__links`. It +represents the `links` property of the `content` resource. The links table contains the +`JSON_parentId` column, which includes a generated hash, such as +`campaigns/{campaign_id}/content_1c3b951ece2a05c1239b06e99cf804c2`, whose value is inserted into +the `links` column of the campaign content table. This is done automatically because once +you say that the entire response is supposed to be a single table row, the array `_links` +property will not fit into a single value of a table. + +## Summary +Now that you know how to extract sub-resources using child jobs, as well as resources composed directly of +properties (without an array of items), you probably think that the `_links` property, found all over the +MailChimp API and giving us a lot of trouble, is best to be ignored. The answer to this is +**mapping**, described in the tutorial's [next part](/extend/generic-extractor/tutorial/mapping/). + +You might also have noticed some duplicate records in the table `in.c-ge-tutorial.campaigns__campaign_id__content` +along the way. You'll look into this as well. diff --git a/src/content/docs/extend/generic-extractor/tutorial/json/index.md b/src/content/docs/extend/generic-extractor/tutorial/json/index.md new file mode 100644 index 000000000..afe509a16 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/tutorial/json/index.md @@ -0,0 +1,145 @@ +--- +title: JSON Introduction +slug: 'extend/generic-extractor/tutorial/json' +--- + + +[JSON (JavaScript Object Notation)](http://www.json.org/) is an easy-to-work-with format for describing structured +data. Before you start working with JSON, familiarize yourself with basic programming jargon. It is also recommended +to have a text editor with JSON support (you can also use an [online editor](http://www.jsoneditoronline.org/)). + +## Object Representation +To describe structured data, JSON uses **objects** and **arrays**. + +### Objects +Objects consist of **properties** and their **values**. Because the values in an object are identified by names +(property names), they are not kept in a particular order. + +The following object describes *John Doe* using two properties : `firstName` and `lastName`. + +```json +{ + "firstName": "John", + "lastName": "Doe" +} +``` + +*Notice that the object is enclosed in `{}`. The properties and values are both in double quotes and are separated +by a colon. The individual properties are separated from each other using commas.* + +### Arrays +As objects collect named values, **arrays** are ordered lists of values that do not have a property +name but are identified by their numeric position. + +Let's go on to describing John Doe's family using an **array** (marked by `[]`) of three **objects**: + +```json +[ + { + "firstName": "John", + "lastName": "Doe", + "role": "father" + }, + { + "firstName": "Jenny", + "lastName": "Doe", + "role": "mother" + }, + { + "firstName": "Jimmy", + "lastName": "Doe", + "role": "son" + } +] +``` + +*Objects are also separated from each other by commas. Notice that the last item (property or object) is not +followed by a comma.* + +### Terminology +The terminology varies a lot and other expressions are also commonly used: + +- Object --- also a record / structure / dictionary / hash table / keyed list / key value / associative array +- Property --- also a field / key / index +- Array --- also a collection / list / vector / ordinal array / sequence + +## Data Values +Each property value always has one of the following data types: + +- String --- text +- Number --- number +- Integer --- whole number (without decimal part) +- Boolean --- value which is either `true` or `false` +- Array --- ordered list of values +- Object --- collection of named values + +The types `string`, `number`, `integer` and `boolean` represent **scalar values** (simple). The types `array` and +`object` represent **structured values** (they are composed of other values). For example: + +```json +{ + "stringProperty": "someText", + "numberProperty": 12.45, + "integerProperty": 42, + "booleanProperty": false, + "arrayProperty": ["first", "second"], + "objectProperty": { + "name": "John", + "surname": "Doe" + } +} +``` + +*Notice that only strings and property names are enclosed in double quotes. The boolean value is `false` without +double quotes because `false`, `true` and `null` (no or an unknown value) are **keywords**, not strings.* + +## References +There are multiple ways to refer to particular properties in a JSON document (for instance, [JSONPath](http://jsonpath.com/). +For the purpose of this documentation, we will use simple *dot notation*. Let's consider this JSON describing the +Doe's family: + +```json +{ + "address": { + "city": "Fresno", + "street": "Main Street" + }, + "members": [ + { + "firstName": "John", + "age": 42, + "shoeSize": 42.5, + "lastName": "Doe", + "interests": ["cars", "girls", "lego"], + "adult": true + }, + { + "firstName": "Jenny", + "adult": true, + "shoeSize": 24.5, + "lastName": "Doe", + "age": 42, + "interests": ["cars", "boys", "painting"] + }, + { + "adult": false, + "firstName": "Jimmy", + "lastName": "Doe", + "shoeSize": null, + "age": 1, + "interests": ["cars", "lego", "painting"] + } + ] +} +``` + +To refer to John's city, we would write `address.city`. To refer to little Jimmy's shoe size, we +would write `members[2].shoeSize`. Array items indexes are *zero-based*, so the third item has +index `2`. + +The order of items in an object is not important. It is also worth noting that `[]` represents an empty array and +`{}` represents an empty object. + +## Summary +This page contains a little introduction to JSON documents. We intentionally avoided many details, +but you should now understand what JSON is, and how to write some stuff in it. diff --git a/src/content/docs/extend/generic-extractor/tutorial/mailchimp-api-docs-1.png b/src/content/docs/extend/generic-extractor/tutorial/mailchimp-api-docs-1.png new file mode 100644 index 000000000..8e61aaf81 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/mailchimp-api-docs-1.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/mailchimp-api-docs-2.png b/src/content/docs/extend/generic-extractor/tutorial/mailchimp-api-docs-2.png new file mode 100644 index 000000000..1ae66b621 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/mailchimp-api-docs-2.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/mapping/index.md b/src/content/docs/extend/generic-extractor/tutorial/mapping/index.md new file mode 100644 index 000000000..b52ad71c1 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/tutorial/mapping/index.md @@ -0,0 +1,350 @@ +--- +title: Mapping Tutorial +slug: 'extend/generic-extractor/tutorial/mapping' +--- + + +In the previous part of the tutorial, you [extracted the content of a MailChimp campaign](/extend/generic-extractor/tutorial/jobs/). +Now, it's time to clean up the response. + +This is the initial configuration: + +![Whole cfg](/extend/generic-extractor/tutorial/mapping_all.png) + +**In JSON:** + +```json +{ + "parameters": { + "api": { + "baseUrl": "https://us13.api.mailchimp.com/3.0/", + "authentication": { + "type": "basic" + }, + "pagination": { + "method": "offset", + "offsetParam": "offset", + "limitParam": "count", + "limit": 1 + } + }, + "config": { + "debug": true, + "username": "dummy", + "#password": "c40xxxxxxxxxxxxxxxxxxxxxxxxxxxxx-us13", + "outputBucket": "ge-tutorial", + "jobs": [ + { + "endpoint": "campaigns", + "dataField": "campaigns", + "children": [ + { + "endpoint": "campaigns/{campaign_id}/send-checklist", + "dataField": { + "path": "items", + "delimiter": "." + }, + "placeholders": { + "campaign_id": "id" + } + }, + { + "endpoint": "campaigns/{campaign_id}/content", + "dataField": { + "path": ".", + "delimiter": "." + }, + "placeholders": { + "campaign_id": "id" + } + } + ] + } + ] + } + } +} +``` + +It extracts MailChimp campaigns with the `send-checklist` items and campaign `content`. +However, you are probably not interested in some parts of the content resource. Also, the table contains duplicates. + +***Technical note on duplicates:** If you examine the job events, you will see +the request `GET /3.0/campaigns/f7ed43aaea/content?count=1&offset=0` sent. That is, +pagination applies to **all API requests**. Generic Extractor tries to page the +unpaged `/content` resource. This may ultimately lead to duplicates because the extraction of that +resource is only terminated after the resource returns the same response twice.* + +## Mapping + +A mapping defines the shape of Generic Extractor outputs. It is stored +in the `config.mappings` property and is identified by the resource data type. +When a resource is assigned an internal `Result Name` (`dataType`), a mapping can be created +for it. To use a mapping, first define a `Result Name` (`dataType`) in the job property. + +### UI + +The mapping can be created in the `Mapping section` in the UI by clicking the `Create Mapping` toggle. + +![Create mapping](/extend/generic-extractor/tutorial/create_mapping_toggle.png) + +You may generate the mapping automatically by clicking the**Infer Mapping** button in the top right corner. + +This operation will generate a mapping based on the sample response of the endpoint. + +![Create mapping](/extend/generic-extractor/tutorial/create_mapping.png) + +#### Primary key +To create a primary key, you can specify a `.` separated path of the elements in the response. ***Note:** If you are mapping child jobs, +the parent keys will automatically be included.* + +#### Nesting level +Currently, the automatic detection outputs only single table mapping. You can control the nesting level by specifying +the `Nesting Level` property. For example, a depth of 1 transforms `{"address": {"street": "Main", "details": {"postcode": "170 00"}}}` into two columns: +`address_street` and `address_details`. +All elements that have ambiguous types or are beyond the specified depth are stored in a single column as JSON, e.g., with the [`force_type`](/extend/generic-extractor/configuration/config/mappings/#mapping-without-processing) option. + +For example, if you click to generate mapping on the `Campaigns` endpoint with level 2 and primary key `id`, you will get this result +(note the link between the `Result Name` (`dataType`) and mappings key): + +``` +"mappings": {"campaigns": { + "id": { + "mapping": { + "destination": "id", + "primaryKey": true + } + }, + "web_id": "web_id", + "type": "type", + "create_time": "create_time", + "archive_url": "archive_url", + "long_archive_url": "long_archive_url", + "status": "status", + "emails_sent": "emails_sent", + "send_time": "send_time", + "content_type": "content_type", + "needs_block_refresh": "needs_block_refresh", + "resendable": "resendable", + "recipients.list_id": "recipients_list_id", + "recipients.list_is_active": "recipients_list_is_active", + "recipients.list_name": "recipients_list_name", + "recipients.segment_text": "recipients_segment_text", + "recipients.recipient_count": "recipients_recipient_count", + "settings.subject_line": "settings_subject_line", + "settings.title": "settings_title", + "settings.from_name": "settings_from_name", + "settings.reply_to": "settings_reply_to", + "settings.use_conversation": "settings_use_conversation", + "settings.to_name": "settings_to_name", + "settings.folder_id": "settings_folder_id", + "settings.authenticate": "settings_authenticate", + "settings.auto_footer": "settings_auto_footer", + "settings.inline_css": "settings_inline_css", + "settings.auto_tweet": "settings_auto_tweet", + "settings.fb_comments": "settings_fb_comments", + "settings.timewarp": "settings_timewarp", + "settings.template_id": "settings_template_id", + "settings.drag_and_drop": "settings_drag_and_drop", + "tracking.opens": "tracking_opens", + "tracking.html_clicks": "tracking_html_clicks", + "tracking.text_clicks": "tracking_text_clicks", + "tracking.goal_tracking": "tracking_goal_tracking", + "tracking.ecomm360": "tracking_ecomm360", + "tracking.google_analytics": "tracking_google_analytics", + "tracking.clicktale": "tracking_clicktale", + "delivery_status.enabled": "delivery_status_enabled", + "_links": { + "type": "column", + "mapping": { + "destination": "links" + }, + "forceType": true + } + +}} +``` + +### JSON + +The value of the `Result Name` (`dataType`) property is an arbitrary name. Apart from identifying +the resource type, it is also used as the **output table name**. If you run +the job, the content will be stored in `in.c-ge-tutorial.content`. + +Each mapping item is identified by the property name of the resource and must contain +`mapping.destination` with the target column name in the output table. For example: + +```json +"mappings": { + "content": { + "plain_text": { + "mapping": { + "destination": "text" + } + } +``` + +The above mapping setting defines that the +resource property `plain_text` will be stored in the table column `text` for the `content` data type. No other +properties of the content resource will be imported. In other words, the mapping defines +all columns of the output table. + +To give an example, if you are interested in having the `plain_text` and `html` versions of the +campaign content, use a mapping like this: + +```json +"mappings": { + "content": { + "plain_text": { + "mapping": { + "destination": "text" + } + }, + "html": { + "mapping": { + "destination": "html" + } + } + } +} +``` + +Note that the `destination` value is arbitrary but must be a valid column name. +The data type name (`content`) must match the value of the `dataType` property +as defined in some jobs. + +## Parent Reference +The above mapping works, but it is missing the campaign ID, and you cannot +match the content to some campaign records. Therefore, you must extract the campaign ID +from the context (i.e., the job parameter). This can be done using a special `user` mapping. + +When the mapping `type` is set to `user`, use the special prefix `parent_` to refer to +a `placeholder` defined in the job. You can create the following mapping: + +```json +"mappings": { + "content": { + "parent_id": { + "type": "user", + "mapping": { + "destination": "campaign_id" + } + } + } +} +``` + +The above configuration defines a mapping for the `content` data type. +In the result table named `content`, the column `campaign_id` will be created. +Its content will be the value of the `id` placeholder +(`parent_id` minus the `parent_` prefix) in the respective job. + +Apart from specifying what columns should be in the output table, the +mapping allows you to set a column as part of a primary key. The entire configuration would +then look like this: + +```json +{ + "parameters": { + "api": { + "baseUrl": "https://us13.api.mailchimp.com/3.0/", + "authentication": { + "type": "basic" + }, + "pagination": { + "method": "offset", + "offsetParam": "offset", + "limitParam": "count", + "limit": 10 + } + }, + "config": { + "debug": true, + "username": "dummy", + "#password": "c40xxxxxxxxxxxxxxxxxxxxxxxxxxxxx-us13", + "outputBucket": "ge-tutorial", + "jobs": [ + { + "endpoint": "campaigns", + "dataField": "campaigns", + "children": [ + { + "endpoint": "campaigns/{campaign_id}/send-checklist", + "dataField": "items", + "placeholders": { + "campaign_id": "id" + } + }, + { + "endpoint": "campaigns/{campaign_id}/content", + "dataField": ".", + "dataType": "content", + "placeholders": { + "campaign_id": "id" + } + } + ] + } + ], + "mappings": { + "content": { + "parent_id": { + "type": "user", + "mapping": { + "destination": "campaign_id", + "primaryKey": true + } + }, + "plain_text": { + "mapping": { + "destination": "text" + } + }, + "html": { + "mapping": { + "destination": "html" + } + } + } + } + } + } +} +``` + +## Review +Now, let's review what parts are connected and how. Note that the values in blue +have been chosen arbitrarily when the configuration was created: + +![Configuration Schema](/extend/generic-extractor/tutorial/configuration-schema.svg) + +## Summary +Mapping lets you precisely define what the extraction output will look like; it also +defines primary keys. + +If you do a one-time ad-hoc extraction, you may skip setting up the mapping and clean +the extracted data later in [Transformations](/transformations/). +However, if you intend to use your configuration regularly or want to make it into a component, +setting up a mapping is recommended. + +## Tips and Tricks + +### Key Containing a Dot Character + +The key of the mapping supports dot notation to traverse into children. So, if the key contains a dot, you need to change the delimiter. See the following example: + +```json +"mappings": { + "content": { + "created.date": { + "delimiter": "/", + "type": "column", + "mapping": { + "destination": "createdDate" + } + } + } +} +``` + +As you changed the delimiter from the default `.` to `/`, it's no longer parsed as two separate keys `created` and `date`, but rather just a single key `created.date`. diff --git a/src/content/docs/extend/generic-extractor/tutorial/mapping_all.png b/src/content/docs/extend/generic-extractor/tutorial/mapping_all.png new file mode 100644 index 000000000..f7b207ed3 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/mapping_all.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/new_endpoint.png b/src/content/docs/extend/generic-extractor/tutorial/new_endpoint.png new file mode 100644 index 000000000..aeec1bfba Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/new_endpoint.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/new_endpoint_modal.png b/src/content/docs/extend/generic-extractor/tutorial/new_endpoint_modal.png new file mode 100644 index 000000000..6425aaa54 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/new_endpoint_modal.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/pagination.png b/src/content/docs/extend/generic-extractor/tutorial/pagination.png new file mode 100644 index 000000000..3b892a17e Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/pagination.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/pagination/index.md b/src/content/docs/extend/generic-extractor/tutorial/pagination/index.md new file mode 100644 index 000000000..1351a7f64 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/tutorial/pagination/index.md @@ -0,0 +1,165 @@ +--- +title: Pagination Tutorial +slug: 'extend/generic-extractor/tutorial/pagination' +--- + + +Pagination breaks a result with a large number of items into separate pages and is used very commonly in +many API calls. + +In the previous part of the tutorial, you [fetched campaigns from +the MailChimp API](/extend/generic-extractor/tutorial/). If you created a new account, chances are that you probably have +only one campaign. You should now create some more campaigns (you do not have to configure them anyhow). + +If the API has consistent pagination for all resources (which the +[MailChimp API has](https://mailchimp.com/developer/guides/get-started-with-mailchimp-api-3/#Parameters)), +then the pagination is defined in the `Pagination` section of the endpoint configuration (or in the `api` section in the underlying JSON). + +## Preparation +The MailChimp API uses the [`offset` pagination method](https://mailchimp.com/developer/guides/get-started-with-mailchimp-api-3/#Parameters), +which means that each page has a fixed `limit` (by default 10 items), and you need to use the offset to move +that fixed-size page over the next set of results. For the first page, the `offset` is 0, for the second +page, the `offset` is 10. This is the same kind of pagination as in SQL. + +The offset pagination method is configured with the following basic properties: + +- `method` --- for MailChimp, set this property to `offset`. +- `offsetParam` --- name of the API parameter which defines the [page offset](https://mailchimp.com/developer/guides/get-started-with-mailchimp-api-3/#Parameters) +- `limitParam` -- name of the API parameters which define the [page size (limit)](https://mailchimp.com/developer/guides/get-started-with-mailchimp-api-3/#Parameters) + +So, for MailChimp, configure the pagination this way: + +- Click `Create New Pagination` in the Endpoint's Pagination section: + +![Create pagination.png](/extend/generic-extractor/tutorial/img.png) + +- Name your pagination and select the Offset method: + +![Pagination](/extend/generic-extractor/tutorial/pagination.png) + +### JSON + +The resulting JSON configuration will look like this: + +```json +"api": { + "baseUrl": "https://us13.api.mailchimp.com/3.0/", + "authentication": { + "type": "basic" + }, + "pagination": { + "method": "multiple", + "scrollers": { + "default": { + "method": "offset", + "limit": 100, + "limitParam": "count", + "offsetParam": "offset", + "firstPageParams": true, + "offsetFromJob": false + } + } + } +} + +``` + +Alternatively, you can use a single pagination method instead of a scroller when configuring manually: + +```json +"api": { + "baseUrl": "https://us13.api.mailchimp.com/3.0/", + "authentication": { + "type": "basic" + }, + "pagination": { + "method": "offset", + "offsetParam": "offset", + "limitParam": "count" + } +}, +``` + +The entire Generic Extractor configuration will look like this: + +```json +{ + "api": { + "baseUrl": "https://us13.api.mailchimp.com/3.0/", + "authentication": { + "type": "basic" + }, + "pagination": { + "method": "multiple", + "scrollers": { + "default": { + "method": "offset", + "limit": 100, + "limitParam": "count", + "offsetParam": "offset", + "firstPageParams": true, + "offsetFromJob": false + } + } + } + }, + "config": { + "outputBucket": "ge-tutorial", + "incrementalOutput": false, + "jobs": [ + { + "__NAME": "campaigns", + "endpoint": "campaigns", + "method": "GET", + "dataType": "campaigns", + "dataField": { + "path": ".", + "delimiter": "." + } + } + ], + "__AUTH_METHOD": "basic", + "username": "dummy", + "#password": "c40xxxxxxxxxxxxxxxxxxxxxxxxxxxxx-us13" + } +} +``` + +***Note:** The `__` prefixed parameters are for internal use by the UI and should not be modified. +Also, they have no effect on component functionality.* + +## Running + +Now, make sure that you have more than one campaign in your account. + +## Testing +Because you probably have fewer than ten (the default page size) campaigns in your MailChimp account, +there is no way to tell whether the pagination works. Let's make sure by setting the `limit` +to 1 and turning the `debug` mode on so that you can see all the requests sent by Generic Extractor. + +Run the configuration and review the events produced by the job. You should see something like this: + +![Screenshot - Debug Events](/extend/generic-extractor/tutorial/job-2.png) + +The oldest events are at the bottom, so you can see that the extractor started by sending an HTTP request: + + GET /3.0/campaigns/?count=1&offset=0 + +Then, it continued with + + GET /3.0/campaigns/?count=1&offset=1 + GET /3.0/campaigns/?count=1&offset=2 + +and so on. You should also see a warning that the `dataField 'campaigns' contains no data`. +This is expected because Generic Extractor tries bigger offsets until the number of returned items is +less than the page size. With the page size set to 1, this means that the last page will contain no data. + +## Summary +In this part of the tutorial, you learned how to set up simple pagination. This is very important +because most APIs use some sort of pagination and without proper setting you would be +getting incomplete data. The next two parts of our tutorial deal with setting up jobs and mapping: + +- [Jobs](/extend/generic-extractor/tutorial/jobs/) --- describe the API endpoints + (resources) to be extracted. +- [Mapping](/extend/generic-extractor/tutorial/mapping/) --- describes how the JSON + response is converted into CSV files that will be imported into Storage. diff --git a/src/content/docs/extend/generic-extractor/tutorial/rest/index.md b/src/content/docs/extend/generic-extractor/tutorial/rest/index.md new file mode 100644 index 000000000..8b5e8c188 --- /dev/null +++ b/src/content/docs/extend/generic-extractor/tutorial/rest/index.md @@ -0,0 +1,134 @@ +--- +title: REST HTTP API Introduction +slug: 'extend/generic-extractor/tutorial/rest' +--- + + +An [API (Application Programming Interface)](https://en.wikipedia.org/wiki/Application_programming_interface) is +an [interface](https://en.wikipedia.org/wiki/Interface_(computing)) to an application, or a **service** +designed for machine access. It can be seen as the UI (User Interface) of an application designed +for machines (other applications). + +So that another application can be programmed to consume the API, it has to have some sort of specification. +A common specification for communicating on the web is the [HTTP protocol](https://en.wikipedia.org/wiki/Hypertext_Transfer_Protocol). +Used by web browsers and other API clients, it defines how two parties (client and server) ought to communicate: + +- Client creates an HTTP **request** and sends it to the server over the network. +- Server processes the request, creates a **response**, and sends it to the client over the network. + +## HTTP Request +An HTTP request is composed of: + +- URL +- HTTP Method +- HTTP Headers +- Optional Body + +### URL +A [URL (Uniform Resource Locator)](https://en.wikipedia.org/wiki/URL) is the address you see in your web browser +address bar. It allows you to **locate a resource**. Each URL has several parts, and it is important to know them. +For example, the address + + https://www.example.com:8080/customers/acme/order/?show=deleted&fields=all + +is composed of: + +- `https` --- **Protocol** (HTTP or HTTPS), +- `www.example.com` --- **Host** --- network address of the HTTP server, +- `8080` --- **port** --- Optional network identifier within the target server; its default value is `80`. +- `/customers/acme/order/` --- Optional **path** to a **resource** we wish to obtain; its default value is `\`. +- `show=deleted&fields=all` --- Optional **request parameters** (also called **query string** or **query +string parameters**), separated by the character `&` (ampersand); the actual parameters are: + - `show` with the value `deleted`, and + - `fields` with the value `all`. + +Because the URL contains a number of special characters (`?`, `&`, `/` and many others), when these parameters +need to be part of the URL, they must be encoded (URL encoded, urlencoded, escaped). Therefore a URL: + + http://example.com/this address & special + +will be actually sent to the server as: + + http%3A%2F%2Fexample.com%2Fthis+address+%26+special + +The web browser (and Generic Extractor too) will normally do this conversion for you. However, you might run into +the encoded format in Generic Extractor events. There are plenty of [online tools to decode](https://urldecode.org/) +this encoded format. + +Sometimes, you may also encounter the term [URI (Uniform Resource Identifier)](https://en.wikipedia.org/wiki/Uniform_Resource_Identifier). +It is used when a single **Resource** may be accessed through multiple URLs. For example, the web page +`http://example.com` may display the same content as `http://example.com/home`. In such case one of the URLs +(probably the second one) is chosen as an identifier, and becomes URI. For our use, there is no important +difference between URI and URL. + +An API **end-point** is identified by its URL, or URI, and should represent a distinct **resource** (users, +invoices etc.). ***Important:** The terms end-point, resource, URL and URI are used interchangeably throughout the +tutorial because they ultimately refer to the same thing.* + +### Method +An HTTP **Method** describes a type of the request to make. It also called an **HTTP Verb** because it +describes what to do with the **resource**. Common HTTP verbs are: + +- `GET` --- Obtain a resource. +- `POST` and `PATCH` --- Update a resource. +- `PUT` --- Create a resource. +- `DELETE` --- Delete a resource. + +Since Generic Extractor only reads data from another API, you will mostly use the `GET` method (and sometimes the +`POST` method). The other HTTP methods are not important for us. + +### Headers +An HTTP request can contain [**headers**](https://en.wikipedia.org/wiki/List_of_HTTP_header_fields#Request_Headers), +which include additional information about the request and response. A typical example of a header is +`Content-type`. For instance, for a web page, `Content-Type: text/html` would be used because an +[HTML page](https://en.wikipedia.org/wiki/HTML) is being transferred. For an API request, it is commonly set +to `Content-type: application/json` because we are transferring [JSON data](http://www.json.org/). + +Apart from standard headers, there are also non-standard headers; these are marked with the prefix `X-`. An +example is the `X-StorageAPIToken` header used with Keboola [Storage API](/integrate/storage/api/). + +### Body +The `POST`, `PUT` and `PATCH` requests can send parameters the same way as the `GET` requests in the URL. +But they can also send them in the request **body**. These are sometimes called **POST data/postdata**. + +## HTTP Response +An HTTP response is composed of: + +- Response Headers --- same as the request headers (only sent by the server) +- Response Body --- actual content of the resource +- Status Code --- status of the request + +#### HTTP Status +The HTTP Status and [status code](https://en.wikipedia.org/wiki/List_of_HTTP_status_codes) represent +a standardized way of describing the response state. For example, the status `200 OK` (200 is the status code) +is associated with a successful response. There are many HTTP Statuses, but the following rules apply: + +- Status codes `2xx` (e.g., 200) represent success. +- Status codes `3xx` represent [redirection](https://en.wikipedia.org/wiki/URL_redirection). +- Status codes `4xx` represent a client error (the request is wrong). +- Status codes `5xx` represent a server error (the server failed to create the response). + +## REST API +[REST (Representational state transfer)](https://www.restapitutorial.com/lessons/whatisrest.html) (or RESTful) +is an API which follows a set of [loosely defined](http://restcookbook.com/Miscellaneous/rest-and-http/) principles: + +- The API URLs (or URIs) represent individual **resources**. Each API endpoint should represent a resource of +a *single type*. For example, it represents a list of users, and not a list of users and their invoices. +- Each resource is **represented** in a structured format ([JSON](http://www.json.org/) or +[XML](https://en.wikipedia.org/wiki/XML)). The data is not transferred, for instance, as ordinary text or a web +page. +- **Messages** (request and response) are transferred using various HTTP methods (`GET`, `POST`, etc.). +For example, for obtaining data, the `GET` method should be used. Also the `GET` method +should not cause any modifications of data. +- The entire communication is **stateless**. This means that multiple requests can be called in an +arbitrary order and must yield the same results. It is not correct for an API to have endpoints such as +`setFilter` and`getFilteredResult` because they imply that any state (a filter) is retained between those API +endpoints. + +## Summary +The above describes the basic concepts of an API, HTTP protocol and HTTP REST API. When you +understand these concepts (and the associated jargon), you can use Generic Extractor +to get responses from virtually any HTTP REST API. Since the REST rules are not rigidly specified, it +is not possible to ensure that Generic Extractor will be capable of reading 100% of APIs, +even when declared as RESTful by someone. + diff --git a/src/content/docs/extend/generic-extractor/tutorial/sub-resources-docs.png b/src/content/docs/extend/generic-extractor/tutorial/sub-resources-docs.png new file mode 100644 index 000000000..2ec676187 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/sub-resources-docs.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/table-campaigns-sample.png b/src/content/docs/extend/generic-extractor/tutorial/table-campaigns-sample.png new file mode 100644 index 000000000..fd906395e Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/table-campaigns-sample.png differ diff --git a/src/content/docs/extend/generic-extractor/tutorial/test_endpoint.png b/src/content/docs/extend/generic-extractor/tutorial/test_endpoint.png new file mode 100644 index 000000000..64cd43881 Binary files /dev/null and b/src/content/docs/extend/generic-extractor/tutorial/test_endpoint.png differ diff --git a/src/content/docs/extend/generic-extractor/ui.png b/src/content/docs/extend/generic-extractor/ui.png new file mode 100644 index 000000000..6cb30b30b Binary files /dev/null and b/src/content/docs/extend/generic-extractor/ui.png differ diff --git a/src/content/docs/extend/generic-writer/configuration-examples/index.md b/src/content/docs/extend/generic-writer/configuration-examples/index.md new file mode 100644 index 000000000..756513286 --- /dev/null +++ b/src/content/docs/extend/generic-writer/configuration-examples/index.md @@ -0,0 +1,297 @@ +--- +title: Generic Writer Configuration Examples +slug: 'extend/generic-writer/configuration-examples' +--- + + +### Configuration Example – Iterations + +This configuration sends the POST request to `https://example.com/test/[[id]]` where `[[id]]` is a column expected in the input table. +It will send as many requests as there are rows in the input table. Each request object is wrapped in `{"data":{}}` object. + +```json +{ + "api": { + "base_url": "https://example.com" + }, + "user_parameters": { + "date": { + "function": "concat", + "args": [ + { + "function": "string_to_date", + "args": [ + "yesterday", + "%Y-%m-%d" + ] + }, + "T" + ] + } + }, + "request_parameters": { + "method": "POST", + "endpoint_path": "/test/[[id]]?", + "headers": { + "Authorization": { + "attr": "token_encoded" + }, + "Content-Type": "application/json" + }, + "query_parameters": { + "date": { + "attr": "date" + } + } + }, + "request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "_", + "chunk_size": 1, + "column_data_types": { + "autodetect": true + }, + "request_data_wrapper": "{ \"data\": [[data]]}", + "column_names_override": {} + }, + "iterate_by_columns": [ + "id" + ] + } +} +``` + +### Exponea Batch Events Writer + +Write customer [events](https://docs.exponea.com/reference#add-event) into the [Exponea API](https://docs.exponea.com) +in [batches](https://docs.exponea.com/reference#batch-commands) of `3` requests. + +**Writer config:** + +```json +{ + "debug": false, + "api": { + "base_url": "https://api-demoapp.exponea.com" + }, + "user_parameters": { + "#token": "12345", + "token_encoded": { + "function": "concat", + "args": [ + "Basic ", + { + "function": "base64_encode", + "args": [ + { + "attr": "#token" + } + ] + } + ] + } + }, + "request_parameters": { + "method": "POST", + "endpoint_path": "/track/v2/projects/1234566/batch?", + "headers": { + "Authorization": { + "attr": "token_encoded" + }, + "Content-type": "application/csv" + } + }, + "request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "__", + "chunk_size": 3, + "column_data_types": { + "autodetect": true + }, + "request_data_wrapper": "{\"commands\":{{data}}}" + } + } +} +``` + +**Input table:** + +| name | data__customer_ids__registered | data__properties__price | data__timestamp | data__event_type | data__properties__test | +|------------------|--------------------------------|-------------------------|-----------------|------------------|------------------------| +| customers/events | milan@test.com | 150 | 123456.78 | testing_event | a | +| customers/events | petr@test.com | 150 | 123456.78 | testing_event | a | +| customers/events | masha@test.com | 150 | 123456.78 | testing_event | a | + +**Result request:** + +```json +{ + "commands": [{ + "name": "customers/events", + "data": { + "customer_ids": { + "registered": "milan@keboola.com" + }, + "properties": { + "price": 150, + "test": "a" + }, + "timestamp": 123456.78, + "event_type": "testing_event" + } + }, { + "name": "customers/events", + "data": { + "customer_ids": { + "registered": "petr@keboola.com" + }, + "properties": { + "price": 150, + "test": "a" + }, + "timestamp": 123456.78, + "event_type": "testing_event" + } + }, { + "name": "customers/events", + "data": { + "customer_ids": { + "registered": "masha.reutovski@keboola.com" + }, + "properties": { + "price": 150, + "test": "a" + }, + "timestamp": 123456.78, + "event_type": "testing_event" + } + } + ] +} + +``` + +### Customer.io User Event + +Update user events via the [Customer.io API](https://customer.io/docs/api/#apitrackeventsevent_add) based on the user_id column. + +The API uses Basic http authentication. + +**Writer config:** + +```json +{ + "api": { + "base_url": "https://track.customer.io", + "authentication": { + "type": "BasicHttp", + "parameters": { + "username": "test_user", + "#password": "pass" + } + } + }, + "request_parameters": { + "method": "POST", + "endpoint_path": "/api/v1/customers/{{user_id}}/events?", + "headers": { + "Authorization": { + "attr": "token_encoded" + }, + "Content-type": "application/csv" + } + }, + "request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "_", + "chunk_size": 1, + "column_data_types": { + "autodetect": true + }, + "request_data_wrapper": "", + "column_names_override": {} + }, + "iterate_by_columns": [ + "user_id" + ] + } +} +``` + +**Input Table:** + +| user_id | data_price | data_date | name | +|----------------|------------|-----------|---------------| +| a@test.com | 150 | 1.1.20 | testing_event | +| petr@test.com | 150 | 1.1.20 | testing_event | +| masha@test.com | 150 | 1.1.20 | testing_event | + +**Json request:** + +For each row in the input one request: + +POST `https://track.customer.io/api/v1/customers/a@test.com/events` + +```json +{"data": {"price": 150, "date": "1.1.20"}, "name": "testing_event"} +``` + +### Slack Notification + +Send notifications to Slack channels via an API. Note that you need to create an app with appropriate permissions at https://api.slack.com/apps + and retrieve the API token. + +**Input Table:** + +| channel | text | +|----------------|------------| +| AC098098 | Hello | +| AC092131 | World | + +**Configuration:** + +```json +{ + "debug": true, + "api": { + "base_url": "https://slack.com" + }, + "user_parameters": { + "#token": "", + "token_encoded": { + "function": "concat", + "args": [ + "Bearer ", + { + "attr": "#token" + } + ] + } + }, + "request_parameters": { + "method": "POST", + "endpoint_path": "/api/chat.postMessage?", + "headers": { + "Authorization": { + "attr": "token_encoded" + }, + "Content-type": "application/json" + } + }, + "request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "_", + "chunk_size": 1, + "column_data_types": { + "autodetect": true + }, + "request_data_wrapper": "" + } + } +} + +``` diff --git a/src/content/docs/extend/generic-writer/configuration/index.md b/src/content/docs/extend/generic-writer/configuration/index.md new file mode 100644 index 000000000..3c3edc2fe --- /dev/null +++ b/src/content/docs/extend/generic-writer/configuration/index.md @@ -0,0 +1,951 @@ +--- +title: Generic Writer Configuration +slug: 'extend/generic-writer/configuration' +--- + + +This component allows you to write data to a specified endpoint in a specified format. It currently supports single +table and single endpoint per configuration. + +The data can be sent in two ways: + +1. Send all content at once - either BINARY or JSON in chunks +2. [Iterate](/extend/generic-writer/configuration/#iterate-by-columns) through each row - where the data is sent in + iterations specified in the input data. By default 1 row = 1 iteration. This allows to change the endpoint + dynamically based on the input using placeholders: `www.example.com/api/user/{{id}}`. Or sending data with different + user parameters that are present in the input table. + +### Configuration parameters + +*Click on the section names if you want to learn more.* + +- [**api**](/extend/generic-writer/configuration/#api) --- [REQUIRED] sets the basic properties of the API. + - [**base_url**](/extend/generic-writer/configuration/#base-url) --- [REQUIRED] defines the URL to which the API requests + should be sent. + - [**authentication**](/extend/generic-writer/configuration/#authentication) --- needs to be configured for any API + which is not public. + - [**retry_config**](/extend/generic-writer/configuration/#retry-config) --- automatically, and repeatedly, retries + failed HTTP requests. + - [**default_query_parameters**](/extend/generic-writer/configuration/#default-query-parameters) --- sets the + default query parameters sent with each API call. + - [**default_headers**](/extend/generic-writer/configuration/#default-headers) --- sets the default query headers + sent with each API call. + - [**ssl_verification**](/extend/generic-writer/configuration/#ssl-verification) --- allows turning of the SSL certificate + verification. Use with caution. + - [**timeout**](/extend/generic-writer/configuration/#timeout) --- maximum time in seconds for which the component + waits after each request (defaults to None if not set). +- [**user_parameters**](/extend/generic-writer/configuration/#user-parameters) --- user parameters to be used in various + contexts, e.g. passwords. Supports dynamic functions. +- [**request_parameters**](/extend/generic-writer/configuration/#request-parameters) --- [REQUIRED] HTTP parameters of the request + - [**method**](/extend/generic-writer/configuration/#method) --- [REQUIRED] defines the HTTP method of the requests. + - [**endpoint_path**](/extend/generic-writer/configuration/#enpoint-path) --- [REQUIRED] relative path of the endpoint. + - [**query_parameters**](/extend/generic-writer/configuration/#query-parameters) --- query parameters sent with each + request + - [**headers**](/extend/generic-writer/configuration/#headers) --- headers sent with each request +- [**request_content**](/extend/generic-writer/configuration/#request-content) --- [REQUIRED] defines how the data is sent + - [**content_type**](/extend/generic-writer/configuration/#content-type) --- [REQUIRED] defines how the data is transferred ( + JSON, binary file, Empty, etc.) + - [**json_mapping**](/extend/generic-writer/configuration/#json-mapping) --- defines the CSV 2 JSON conversion in + case of JSON content type. + - [**iterate_by_columns**](/extend/generic-writer/configuration/#iterate-by-columns) --- defines set of columns in + the input data that are excluded from the content and may be used instead of placeholders within the + request_options. The input table is iterated row by row, e.g. 1 row = 1 request +- [**debug**](/extend/generic-writer/configuration/#debug) --- Turns on more verbose logging for debugging purposes. + +There are also simple pre-defined [**functions**](/extend/generic-writer/configuration/#dynamic-functions) available, +adding extra flexibility when needed. + +### Configuration Map + +The following sample configuration shows various configuration options and their nesting. You can use the map to +navigate between them. + +```json { + "parameters": { + "debug": false, + "api": { + "base_url": "https://example.com/api", + "default_query_parameters": { + "content_type": "json" + }, + "default_headers": { + "Authorization": { + "attr": "#token" + } + }, + "retry_config": { + "max_retries": 5, + "codes": [ + 500, + 429 + ] + }, + "ssl_verification": true, + "timeout": 5 + }, + "user_parameters": { + "#token": "Bearer 123456", + "date": { + "function": "concat", + "args": [ + { + "function": "string_to_date", + "args": [ + "yesterday", + "%Y-%m-%d" + ] + }, + "T" + ] + } + }, + "request_parameters": { + "method": "POST", + "endpoint_path": "/customer/[[id]]", + "headers": { + "Content-Type": "application/json" + }, + "query_parameters": { + "date": { + "attr": "date" + } + } + }, + "request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "__", + "chunk_size": 100, + "column_data_types": { + "autodetect": true, + "datatype_override": [ + { + "column": "phone", + "type": "string" + }, + { + "column": "rank", + "type": "number" + }, + { + "column": "is_active", + "type": "bool" + } + ] + }, + "request_data_wrapper": "{ \"data\": [[data]]}", + "column_names_override": { + "full_name": "FULL|NAME" + } + }, + "iterate_by_columns": [ + "id" + ] + } + } +} ``` + + + + +## Api + +Defines the basic properties of the API that may be shared for multiple endpoints. Such as authentication, base url, +etc. + +### Base URL + +An URL of the endpoint where the payload is being sent. e.g. `www.example.com/api/v1`. + +**NOTE** May contain placeholders for iterations wrapped in `[[]]`,e.g. ``www.example.com/api/v[[api_version]]``. +But in most cases you would set this up on the `endpoint_path` level. + +The parameter `api_version` needs to be specified in the `user_parameters` or in the source data itself if the column is +set as an iteration parameter column. + +### Retry Config + +Here you can set parameters of the request retry in case of failure. + +- `max_retries` --- Number of maximum retries before failure (DEFAULT `1`) +- `codes` --- List of HTTP codes to retry on, e.g. [503, 429] (DEFAULT `(500, 502, 504)`) +- `backoff_factor` --- backoff factor of the exponential backoff. (DEFAULT `0.3`) + +```json +{ + "api": { + "base_url": "https://example.com/api", + "retry_config": { + "max_retries": 5, + "backoff_factor": 0.3, + "codes": [ + 500, + 429 + ] + } + } +} +``` + +### Default Query Parameters + +Allows you to define default query parameters that are being sent with each request. This is useful for +instance for authentication purposes. This is mostly useful for creating Generic Writer templates and registered +components. + +**NOTE** That you can reference parameters defined in `user_parameters` using the `{"attr":"SOME_KEY"}` syntax. + +```json +{ + "api": { + "base_url": "https://example.com/api", + "default_query_parameters": { + "content_type": "json", + "token": { + "attr": "#token" + } + } + } +} +``` + +### Default Headers + +Allows you to define default query parameters that are being sent with each req This is mostly useful for +creating Generic Writer templates and registered components. + +**NOTE** That you can reference parameters defined in `user_parameters` using the `{"attr":"SOME_KEY"}` syntax. + +```json + { + "api": { + "base_url": "https://example.com/api", + "default_headers": { + "Authorization": { + "attr": "#token" + } + } + } +} +``` + +### Authentication + +Some APIs require authenticated requests to be made. This section allows selecting from predefined auth methods. + +The Authentication object is always in following format: + +```json + +{ + "type": "{SUPPORTED_TYPE}", + "parameters": { + "some_parameter": "test_user" + } +} +``` + +**NOTE** Parameters may be also referenced from the `user_parameters` section using the `{"attr":""}` syntax, +see [example 025](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/025-simple-json-basic-http-auth-from-user-params) + +#### BasicHttp + +Basic HTTP authentication using username and password. + +**Example**: + +```json +"api": { + "base_url": "http://localhost:8000", + "authentication": { + "type": "BasicHttp", + "parameters": { + "username": "test_user", + "#password": "pass" + } + } +} +``` + +See [example 024](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/024-simple-json-basic-http-auth) + +#### BearerToken + +Authorization using the `Bearer token` in the header. E.g. each request will be sent with +header: `"authorization": "Bearer XXXX""` + +**Example**: + +```json +{ + "api": { + "base_url": "http://localhost:8000", + "authentication": { + "type": "BearerToken", + "parameters": { + "#token": "XXXX" + } + } + } +} +``` + +See [example 030](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/030-bearer-token-auth) + +### SSL Verification + +Allows turning of the SSL certificate verification. Use with caution. When set to false, the certificate verification is +turned off. + +```json + +{ + "api": { + "base_url": "http://localhost:8000", + "ssl_verification": false + } +} +``` + +### Timeout + +An optional parameter which allows you to define maximum timeout for each request. If not set, it uses the default requests value: None. + +Possible values: (int, float) + +For more information, refer to [requests docs](https://requests.readthedocs.io/en/stable/user/advanced/#timeouts). + +## User Parameters + +In this section you can defined user parameters to be used in various contexts, e.g. passwords. This is also +the place to use the [dynamic functions](). + +It allows referencing another values from `user_parameters` referenced by `{"attr":"par"}` notation. + +**NOTE** Any parameters prefixed by `#` will be encrypted in the Keboola platform on configuration save. + +```json +{ + "user_parameters": { + "#token": "Bearer 123456", + "date": { + "function": "concat", + "args": [ + { + "function": "string_to_date", + "args": [ + "yesterday", + "%Y-%m-%d" + ] + }, + "T" + ] + } + } +} +``` + +### Referencing parameters + +All parameters defined here can be then referenced using the `{"attr":"PARAMETER_KEY"}` syntax. You may reference them +in the following sections: + +- in the `user_parameters` section itself. +- [`api.default_query_parameters`](/extend/generic-writer/configuration/#default-query-parameters) +- [`api.default_headers`](/extend/generic-writer/configuration/#default-headers) +- [`request_parameters.headers`](/extend/generic-writer/configuration/#headers) +- [`request_parameters.query parameters`](/extend/generic-writer/configuration/#query-parameters) + +See +example [010](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/010-simple-json-user-parameters-various) + +## Request Parameters + +Define parameters of the HTTP request sent. + +### Method + +Request method - POST, PUT, UPDATE, DELETE etc. + +Supported methods: `['GET', 'POST', 'PATCH', 'UPDATE', 'PUT', 'DELETE']` + +```json +"request_parameters": { + "method": "POST", + ... +``` + +### Endpoint path + +A relative path of the endpoint. The final request URL is `base_url` and `endpoint_path` combined. + +e.g. when `base_url` is set to `https://example.com/api` and `endpoint_path` to `/customer` the resulting URL +is `https://example.com/api/customer` + +**NOTE** That it is possible to change the `enpoint_path` dynamically +using [iteration columns](/extend/generic-writer/configuration/#iterate-by-columns) e.g. `/orders/[[id]]` as seen +in [example 005](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/005-json-iterations/) + +```json +{ + "request_parameters": { + "method": "POST", + "endpoint_path": "/customer" + } +} +``` + +### Headers + +Allows you to define default query parameters that are being sent with each request. + +**NOTE** That you can reference parameters defined in `user_parameters` using the `{"attr":"SOME_KEY"}` syntax. + +```json +{ + "request_parameters": { + "method": "POST", + "endpoint_path": "/customer", + "headers": { + "Last-updated": 123343534 + } + } +} +``` + +See [example 006](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/006-simple-json-custom-headers/) + +### Query parameters + +Allows you to define default query parameters that are being sent with each request. + +**NOTE** That you can reference parameters defined in `user_parameters` using the `{"attr":"SOME_KEY"}` syntax. + +```json + { + "request_parameters": { + "method": "POST", + "endpoint_path": "/customer/[[id]]", + "query_parameters": { + "dryRun": true, + "date": { + "attr": "date" + } + } + } +} +``` + +See [example 009](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/009-simple-json-request-parameters/) + +## Request Content + +Defines how to process the input and how the sent content should look like. + +### Content Type + + Defines how the input table is translated to a request: + +- `JSON` - input table is converted into a JSON (see `json_mapping`) sent as `application/json` type. + See [example 001](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/001-simple-json/) +- `JSON_URL_ENCODED` - input table is converted into a JSON and sent as `application/x-www-form-urlencoded`. + See [example 021](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/021-simple-json-url-encoded-form/) +- `BINARY` - input table is sent as binary data (just like `curl --data-binary`). + See [example](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/tests/functional/binary_simple/) +- `BINARY_GZ` - input is sent as gzipped binary data. + See [example](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/tests/functional/binary_gz/) +- `EMPTY_REQUEST` - sends just empty requests. Usefull for triggerring webhooks, DELETE calls, etc. As many requests as + there are rows on the input are sent. Useful with `iterate_by_columns` enabled to trigger multiple endpoints. + See [example 022](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/022-empty-request-iterations-delete/) + +```json + +"request_content": { + "content_type": "JSON", +.... +``` + +### JSON Mapping + +[REQUIRED for JSON based content type] This section defines the CSV 2 JSON conversion in case of JSON content type. + +#### Nesting delimiter + +A string that is used for nesting. e.g. `__`. This way you can define nested objects based on column names. + +e.g. When set to `__` a column value `address__streed` will be converted to `{"address"{"street":"COLUMN_VALUE"}}` + +```json +"request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "_", +... +``` + +See +example [008](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/008-simple-json-nested-object-delimiter/) + +#### Chunk size + +Defines how many rows are being sent in a single request. When set to `1` a single object is sent `{}` ( +see [example 002](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/002-simple-json-chunked-single/)) +, when set to >1 an array of objects is sent `[{}, {}]` ( +see [example 003](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/003-simple-json-chunked-multi/)) + +```json +"request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "_", + "chunk_size": 1, +... +``` + +#### Column datatypes + +Optional configuration of column types. This version supports nesting (three levels) and three datatypes: + +- `bool` - Boolean value case-insensitive conversion: `t`, `true`, `yes`, `1`,`"1"` to `True` and `f`, `false`, `no` + to `False` +- `string` - String +- `number` - Number +- `object` - Object - valid JSON array or JSON object, e.g. ["1","2"], {"key":"val"} + +##### Autodetect + +Default value `true +` +Set this option to `true` to make the parser automatically detect the above datatypes. It may be used in combination +with +`datatype_override` option to force datatype to some columns. + +##### Column datatype override + +[OPTIONAL] + +The `autodetect` option in most cases takes care of the datatype conversion properly. But there are some scenarios where +you want make sure that the datatype conversion is forced. E.g. for `phone_number` column to be treated as String a +mapping should be defined as `"phone_number":"string"`. + +Below are options that can be used as a datatype values: + +if you want the value to be always a string, use `string`, if you want the value to be numeric, use `number`. If you +want it to be Boolean, use `bool` +(case-insensitive conversion: `t`, `true`, `yes` to `True` and `f`, `false`, `no` to `False`) +If the value should be an array or object `object` - valid JSON array or JSON object, e.g. ["1","2"], {"key":"val"} + +**Note** If the `autodetect` option is turned off all unspecified column will be treated as a string. + +```json +{ + "request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "_", + "chunk_size": 1, + "column_data_types": { + "autodetect": true, + "datatype_override": [ + { + "column": "phone", + "type": "string" + }, + { + "column": "rank", + "type": "number" + }, + { + "column": "is_active", + "type": "bool" + } + ] + } + } + } +} +``` + +See [example 007](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/007-simple-json-force-datatype/) + +#### Request Data Wrapper + +[OPTIONAL] + +A wrapper/mask of the parsed data. It needs to be json-encoded json. E.g + +```json +"request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "__", + "chunk_size": 1, + "request_data_wrapper": "{ \"data\": [[data]]}", + ... +} +``` + +Given a single column `user__id` and `chunksize` = 2, the above will cause each request being sent as: + +```json +{ + "data": [ + { + "user": { + "id": 1 + } + }, + { + "user": { + "id": 2 + } + } + ] +} +``` + +See +examples: [012](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/012-simple-json-request-data-wrapper/) + +#### Column names override + +You may override specific column names using the `column_names_override` parameter to be able to generate fields with +characters not supported in Storage column names. + +**NOTE** that this is applied **after** the column type definition, so refer to original name in the `column_types` +config. + +**NOTE2** It is possible to rename nested objects as well. The rename is applied to the leaf node. +E.g. `"address___city":"city.address"` +with delimiter set to `___` will result in `{"address":{"city.address":"SOME_VALUE"}}`. +See [example 23](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/023-simple-json-nested-object-rename-column/) + +**Example:** + +```json +"request_content": { + "content_type": "JSON", + "json_mapping": { + "nesting_delimiter": "_", + "chunk_size": 1, + "column_names_override": { + "field_id": "field-id", + "full_name": "FULL.NAME" + } + } +... +} +``` + +For more details refer to +examples: [20](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/020-simple-json-column-name-override/) +and [23](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/023-simple-json-nested-object-rename-column/) + +### Iterate By Columns + +This parameter allows performing the requests in iterations based on provided parameters within data. The user specifies +columns in the source table that will be used as parameters for each request. The column values may be then used instead +of placeholders within the `request_options`. The input table is iterated row by row, e.g. 1 row = 1 request. + +```json +"request_content": { + "content_type": "JSON", + "iterate_by_columns": [ + "id", "date" + ] +} + +``` + +These will be injected in: + +- `request_parameters.endpoint_path` if placeholder is specified, e.g. `/user/[[id]]` +- `user_parameters` section, any existing parameters with a same name will be replaced by the value from the data. This + allows for example for changing request parameters dynamically `www.example.com/api/user?date=xx` where the `date` + value is specified like: + +```json +{ + "request_parameters": { + "method": "POST", + "endpoint_path": "/customer/[[id]]", + "query_parameters": { + "date": { + "attr": "date" + } + } + } +} + +``` + +**NOTE** The iteration columns may be specified for requests of any content type. The `chunk_size` parameter in JSON +mapping is overridden to `1`. + +See the example configurations: + +- [ex. 005](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/005-json-iterations/) +- Empty request with + iterations [ex. 004](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/004-empty-request-iterations/) + , + [ex. 22](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/022-empty-request-iterations-delete/) +- [ex. 011 placeholders in query parameters](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/011-simple-json-user-parameters-from-iterations/) + +##### Example + +Let's have this table on the input: + +| id | date | name | email | address | +|----|------------|-------|------------|---------| +| 1 | 01.01.2020 | David | d@test.com | asd | +| 2 | 01.02.2020 | Tom | t@test.com | asd | + +Consider following request options: + +```json +{ + "request_parameters": { + "method": "POST", + "endpoint_path": "/user/[[id]]", + "query_parameters": { + "date": { + "attr": "date" + } + } + }, + "request_content": { + "content_type": "JSON", + "iterate_by_columns": [ + "id", + "date" + ] + } +} + +``` + +The writer will run in two iterations: + +**FIRST** With data + +| name | email | address | +|-------|------------|---------| +| David | d@test.com | asd | + +Sent to `www.example.com/api/user/1?date=01.01.2020` + +**SECOND** with data + +| name | email | address | +|-------|------------|---------| +| Tom | t@test.com | asd | + +Sent to `www.example.com/api/user/2?date=01.02.2020` + +## Dynamic Functions + +The application support functions that may be applied on parameters in the configuration to get dynamic values. + +Currently these functions work only in the `user_parameters` scope. Place the required function object instead of the +user parameter value. + +The function values may refer to another user params using `{"attr": "custom_par"}` + +**NOTE:** If you are missing any function let us know or place a PR to +our [repository](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/). It's as simple as adding an +arbitrary method into +the [UserFunctions class](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/src/user_functions.py#lines-7) + +**Function object** + +```json +{ + "function": "string_to_date", + "args": [ + "yesterday", + "%Y-%m-%d" + ] +} +``` + +#### Function Nesting + +Nesting of functions is supported: + +```json +{ + "user_parameters": { + "url": { + "function": "concat", + "args": [ + "http://example.com", + "/test?date=", + { + "function": "string_to_date", + "args": [ + "yesterday", + "%Y-%m-%d" + ] + } + ] + } + } +} + +``` + +#### string_to_date + +Function converting string value into a datestring in specified format. The value may be either date in `YYYY-MM-DD` +format, or a relative period e.g. `5 hours ago`, `yesterday`,`3 days ago`, `4 months ago`, `2 years ago`, `today`. + +The result is returned as a date string in the specified format, by default `%Y-%m-%d` + +The function takes two arguments: + +1. [REQ] Date string +2. [OPT] result date format. The format should be defined as in http://strftime.org/ + +**Example** + +```json +{ + "user_parameters": { + "yesterday_date": { + "function": "string_to_date", + "args": [ + "yesterday", + "%Y-%m-%d" + ] + } + } +} +``` + +The above value is then available in [supported contexts](/extend/generic-writer/configuration/#referencing-parameters) +as: + +```json +"to_date": {"attr": "yesterday_date"} +``` + +#### concat + +Concatenate an array of strings. + +The function takes an array of strings to concatenate as an argument + +**Example** + +```json +{ + "user_parameters": { + "url": { + "function": "concat", + "args": [ + "http://example.com", + "/test" + ] + } + } +} +``` + +The above value is then available in supported contexts as: + +```json +"url": {"attr": "url"} +``` + +#### base64_encode + +Encodes string in BASE64 + +**Example** + +```json +{ + "user_parameters": { + "token": { + "function": "base64_encode", + "args": [ + "user:pass" + ] + } + } +} +``` + +The above value is then available in contexts as: + +```json +"token": {"attr": "token"} +``` + +## Debug + +By setting the root parameter `debug` to `true`, it is possible to enable more verbose logging that will help debugging. + +**CAUTION** Note that higher verbosity causes the writer to print parts of the actual content into the job log, so use with caution. +Always make sure to turn this option off after debugging to prevent any issues. + +```json +{ + "debug": true, + "api": { + "base_url": "http://test.com/api/" + }, + "user_parameters": {}, + "request_parameters": { + "method": "POST", + "endpoint_path": "users/[[id]]" + }, + "request_content": { + "content_type": "BINARY" + } +} +``` diff --git a/src/content/docs/extend/generic-writer/index.md b/src/content/docs/extend/generic-writer/index.md new file mode 100644 index 000000000..12b6a39ad --- /dev/null +++ b/src/content/docs/extend/generic-writer/index.md @@ -0,0 +1,48 @@ +--- +title: Generic Writer +slug: 'extend/generic-writer' +--- + + +Generic Writer is a [Keboola component](/overview/) that allows you to send any type of HTTP requests with or without data to arbitrary HTTP endpoints. + +It is a counterpart to the [Generic Extractor](/extend/generic-extractor) that allows you to extract data from virtually any API. + +The core concepts and configuration of the writer are also quite similar to the Generic Extractor, so it should be easy to +configure for anyone familiar with the extractor. + +## Generic Writer Requirements +The requirements are the same as for the [Generic Extractor](/extend/generic-extractor/#generic-extractor-requirements). +No programming skills or additional tools are required. You just need to do two easy things before you start: + +- Learn how to [write JSON](/extend/generic-extractor/tutorial/json/). +- Have the documentation of your chosen API at hand. + +## Functionality Notes + +The writer writes data to a specified endpoint in a specified format. It supports a single table and a single endpoint per configuration. + +The content can be sent in two ways: + +1. Send all content at once – either BINARY or JSON in chunks +2. Iterate through each row – where the data is sent in iterations specified in the input data. By default 1 row = 1 iteration. +This allows to change the endpoint dynamically based on the input using placeholders: `www.example.com/api/user/{{id}}`. +Or sending data with different user parameters that are present in the input table. + +## Use Cases + +There are variety of use-cases for the generic writer. You may create, update or even delete objects via RESTful API or just trigger +simple webhooks by sending GET requests to specified endpoints or send notifications to slack. The setup is quite straightforward and it +allows you to leverage secure [encripted parameters](overview/encryption/) and dynamic functions. + +**The typical use cases are:** + +- Webhook triggers +- Notifications, e.g., Slack +- Writing JSON data (UPDATES, etc.) +- Sending CSV files as binary data (may be gzipped) +- Calling arbitrary endpoints with parameters defined on the input + - E.g., `DELETE api.com/[[user_id]]` where `user_id` is a column in the input table + +For real configuration examples, see the [configuration examples section](/extend/generic-writer/configuration-examples) + or the collection of [functional examples](https://bitbucket.org/kds_consulting_team/kds-team.wr-generic/src/master/docs/examples/). diff --git a/src/content/docs/extend/index.md b/src/content/docs/extend/index.md new file mode 100644 index 000000000..eacaae8d5 --- /dev/null +++ b/src/content/docs/extend/index.md @@ -0,0 +1,57 @@ +--- +title: Extending Keboola +slug: 'extend' +--- + +As an open system consisting of many built-in, interoperating components, +such as Storage or Extractors, [Keboola](/overview/) can be easily extended. +We encourage you to [**build your own components**](/extend/component/tutorial), whether for +your own use or to be offered to other Keboola users and customers. + +There are two main options for extending Keboola: (a) creating your own **component** and (b) using **Generic +Extractor** to build an extractor for a RESTful API. + +## Advantages of Extending Keboola + +Depending on your role, extending Keboola offers various advantages: + +- If you already are a **Keboola customer**: + - Create your own component to convert your business problem into cloud. We will take care of the technical arrangements around running it. + - Create extractors or writers for communicating with your legacy systems, even if they are completely non-standard. + - Create components to experiment with new business solutions. No need to ask your IT to allocate resources to you. [Fail fast](https://en.wikipedia.org/wiki/Fail-fast#Business). + - Easily access data from many different sources. +- If you are an **external company**: + - Create connectors (Extractors/Writers) so that Keboola users can easily connect to your service and broaden your customer base. + - Create applications containing or using your algorithms and easily "deploy" them to Keboola customers. They won't be exposed to end-users, neither will be the end-user data exposed to you. + - Easily deliver the data back to your customers. +- If you are a **data scientist**: + - Create applications for delivering your work to your customer. We will take care of the technical arrangements. No need to rent servers and feed data to them. + - Make your application or algorithm available to all existing Keboola subscribers and implementation partners. + - Focus only on areas of your product where you are adding value. + - Let Keboola be in charge of the billing. + +## Component +A [component](/extend/component/) can be used as: + +- **Extractor**, allowing customers to get data from new sources. It only processes input tables from external sources (usually API). +- **Application**, further enriching the data or adding value in new ways. It processes input tables stored as CSV files or database tables and generates result tables as CSV files or database tables. +- **Transformation**, allowing customers to modify their data. It is a constrained form of an application. +- **Writer**, pushing data into new systems and consumption methods. It does not generate any data in Keboola projects. +- **Processor**, adjusting the inputs or outputs of other components. It has to be run together with one of the above components. + +All components are run using [Job Queue](/extend/job-queue/), a service that takes +care of their authentication, starting, stopping, isolation, reading data from and writing it to Keboola Storage. They must adhere to the +[common interface](/extend/common-interface/). Creating components requires an elementary knowledge of [Docker](https://www.docker.com/why-docker). +They can be implemented in virtually any programming language and be fully customized and tailored to anyone's needs. +They also support OAuth authorization. To get started with building a component, see our [**tutorial**](/extend/component/tutorial/). + +## Generic Extractor +[Generic Extractor](/extend/generic-extractor/) is a Keboola component acting like a +customizable [HTTP REST client](/extend/generic-extractor/tutorial/rest/). It can be configured to extract data +from virtually any API and offers a vast amount of configuration options. With Generic Extractor, you can build an +entirely new extractor for Keboola in less than an hour. + +Components based on Generic Extractor are built using [JSON configuration](/extend/generic-extractor/tutorial/) and a +[published template](/extend/generic-extractor/publish/). They have a predefined UI, require no knowledge of Docker or +other tools, and they use a Keboola owned [repository](https://github.com/keboola/kbc-ui-templates/). To get +started with Generic Extractor, see our [**tutorial**](/extend/generic-extractor/tutorial/). diff --git a/src/content/docs/extend/job-queue/docker-runner.svg b/src/content/docs/extend/job-queue/docker-runner.svg new file mode 100644 index 000000000..2968f7a7e --- /dev/null +++ b/src/content/docs/extend/job-queue/docker-runner.svg @@ -0,0 +1,2 @@ + +
Job Queue
Job Queue
Isolated Container
Isolated Container
Storage API
Storage API
Project Storage
Project Storage
Job Configuration
[Not supported by viewer]
Storage Input
Storage Input
Storage Output
Storage Output
State
State
Parameters
Parameters
Input Tables
Input Tables
/data/in/tables/
/data/in/files/
[Not supported by viewer]
Output Tables
Output Tables
Output Files
Output Files
/data/out/tables/
/data/out/files/
[Not supported by viewer]
/data/config.json
/data/in/state.json
[Not supported by viewer]
Component Definition
Component Definition
Component Configuration
Component Configuration
Pull Docker Image
Pull Docker Image
stdout / stderr
stdout / stderr
Run Container
Run Container
/data/out/state.json
<div>/data/out/state.json</div>
Events
Events
Input Files
Input Files


<div><br></div><div><br></div>
Job Result
[Not supported by viewer]
Status
Status
Exit code
Exit code
\ No newline at end of file diff --git a/src/content/docs/extend/job-queue/index.md b/src/content/docs/extend/job-queue/index.md new file mode 100644 index 000000000..0dc6bc585 --- /dev/null +++ b/src/content/docs/extend/job-queue/index.md @@ -0,0 +1,90 @@ +--- +title: Job Queue +slug: 'extend/job-queue' +redirect_from: + - /extend/docker-runner/ +--- + + +Job Queue is a core Keboola Service, which +provides an interface for running Keboola components. Every component in Keboola is +represented by a Docker image. +Running a component means creating and executing an [asynchronous job](/integrate/jobs/). + +Developing functionality in [Docker](https://www.docker.com/) allows you to focus only on the application logic; all communication +with the [Storage API](https://api.keboola.com/?service=storage) will be handled by Job Queue. You can encapsulate any application into a Docker image +following a set of rules that will allow you to integrate the application into Keboola. + +There is a [predefined interface](/extend/common-interface/) with Job Queue, consisting +mainly of a [folder structure](/extend/common-interface/folders/) and a [serialized configuration file](/extend/common-interface/config-file/). +All [components](/extend/component/), including our internal R and Python Transformations, are run using Job Queue. + +## Workflow +The Job Queue functionality can be described in the following steps: + +- Download and build the specified Docker image. +- Download all [tables](/extend/common-interface/folders/#dataintables-folder) and [files](/extend/common-interface/folders/#datainfiles-folder) specified in the input mapping from Storage. +- Create a [configuration file](/extend/common-interface/config-file/). +- Run [before processors](/extend/component/processors/) if there are any. +- Run the Docker image (create a Docker container). +- Run [after processors](/extend/component/processors/) if there are any. +- Upload all [tables](/extend/common-interface/folders/#dataouttables-folder) and +[files](/extend/common-interface/folders/#dataoutfiles-folder) in the output mapping to Storage. +- Delete the container and all temporary files. + +When the component execution is finished, Job Queue automatically collects the exit code and the content of STDOUT and STDERR. +The following schema illustrates the workflow of running a dockerized component. + +![Docker Workflow](/extend/job-queue/docker-runner.svg) + +### Features +The component is responsible for these processes: + +- Reading the configuration and source tables in CSV format and files (if specified) +- Writing the results to the predefined folders and files +- Proper handling of success/error results by setting an appropriate exit code + +Job Queue is responsible for the following processes: + +- **Authentication:** Job Queue makes sure the component is run by authorized users/tokens. +It is not possible to run a component anonymously. The component does not have an access to the Keboola token +itself, and it receives only limited information about the project and the end-user. +- **Starting and stopping** the component: Job Queue will boot a Docker container which contains the +component. This ensures the component runs in a precisely defined environment, which is guaranteed to +be the same for each component run. No component state is preserved (with the exception of the +[state file](/extend/common-interface/config-file/#state-file). +- **Reading and writing data** to Keboola Storage: Job Queue ensures a custom component +cannot access arbitrary data in the project. It will only receive the input mapping defined by the end user; +and only those outputs defined in the output mapping by the end user will be written to the project. +- **Component isolation**: Each component is run in its own Docker container, which is isolated from other +containers; the component cannot be affected by other running components. It may also be limited +to have no network access. + +## API +The [Job Queue API](https://api.keboola.com/?service=job-queue) has API calls to + +- run a [component](/extend/component/). +- [encrypt values](/overview/encryption/). +- [prepare the data folder](/extend/component/running/#preparing-the-data-folder). +- run [component actions](/extend/common-interface/actions/). +- run a [component](/extend/component/) with a [specified Docker image tag](https://api.keboola.com/?service=job-queue#post-/jobs), usable for [testing images](/extend/component/deployment/#test-live-configurations). + +## Configuration +Components executed by Job Queue store their configurations in +[Storage API components configurations](https://api.keboola.com/?service=storage#tag--Component-Configurations). + +When creating the configuration, use +[this JSON schema](https://github.com/keboola/docker-bundle/blob/master/Resources/schemas/configuration.json) +to validate the configuration before storing it. The configuration contains the following nodes, +all of them are optional: + +- `parameters` --- an arbitrary object passed to the dockerized application itself +- `storage` --- configuration of [input and output mapping](/extend/common-interface/folders/); specific options correspond to the options of the +[unload data](https://keboola.docs.apiary.io/#reference/tables/unload-data-asynchronously) and +[load data](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async) API calls. +- `runtime` --- [runtime settings](/integrate/jobs/#job-runtime-configuration) (`tag`, `backend`, `parallelism`); most notably `runtime.tag` +pins the Docker image tag that jobs of this configuration run, which is the usual way of testing a development build of a component +- `processors` --- configuration of [Processors](/extend/component/processors/) +- `authorization` --- OAuth authorization [injected to the configuration](/extend/common-interface/oauth/); not stored in the component configuration +- `image_parameters` --- an arbitrary object passed from the [component](/extend/component/); not stored in the component configuration +- `action` --- an [action](/extend/common-interface/actions/) being executed; not stored in the component configuration diff --git a/src/content/docs/extend/publish/approve.png b/src/content/docs/extend/publish/approve.png new file mode 100644 index 000000000..3b9e2bc77 Binary files /dev/null and b/src/content/docs/extend/publish/approve.png differ diff --git a/src/content/docs/extend/publish/checklist/index.md b/src/content/docs/extend/publish/checklist/index.md new file mode 100644 index 000000000..1e8e94ab3 --- /dev/null +++ b/src/content/docs/extend/publish/checklist/index.md @@ -0,0 +1,42 @@ +--- +title: Checklist +slug: 'extend/publish/checklist' +--- + + +This checklist is used for the last check of the component before sending the publish request. +See [Publish Component tutorial](/extend/publish/) for details. + +**Developer Portal** +- The component name doesn't contain words like `extractor`, `application`, and `writer`. +- The component icon is representative and has reasonable quality. It is in `PNG` format and without background. +- The short description describes the **service**, NOT `This extractor extracts ...`. +- Licensing information is valid, and the vendor description is current. +- License and documentation URLs are publicly accessible, no link to a private repository. +- The tag is set to the expected value and uses [semantic versioning](https://semver.org/). +- The correct [data flow is set in UI options](/extend/publish/#component-name-and-description). + +**Component Configuration** +- Sensitive values [use encryption](/overview/encryption/). +- [Configuration and Row schema](/extend/publish/#component-configuration) + - Titles are short and without a colon, period, etc. + - Required properties are listed in the field `required`. + - Each property has defined `propertyOrder`. + - Properties have an explanatory `description` if they are not trivial. +- Configuration description (if used) + - Contains only level 3 `###` and level 4 `####` headers. + - Doesn't repeat what is obvious from Configuration and Row schema. + +**Component Internals** +- Job exits with an understandable [UserError](/extend/common-interface/environment/#return-values) if: + - Empty configuration. + - Invalid credentials. + - Wrong data type used (e.g., string instead of array). + - Missing required property. + - Random/invalid data typed to the configuration properties. + - External server/service is down. + - An expected error occurs (e.g., not found, too many requests, ...). +- Internal messages (e.g., stack trace) with no meaning for the user are not logged. + +**Publication Request** +- A link to the pull request with changes in the [documentation](/) is included (if any). diff --git a/src/content/docs/extend/publish/index.md b/src/content/docs/extend/publish/index.md new file mode 100644 index 000000000..64b22927b --- /dev/null +++ b/src/content/docs/extend/publish/index.md @@ -0,0 +1,124 @@ +--- +title: Publish Component +slug: 'extend/publish' +redirect_from: + - /extend/registration/checklist/ + - /extend/registration/ +--- + + +As described in the [architecture overview](/overview/), Keboola consists of many different components. +Only those components that are published in our **Component List** are generally available in Keboola. +The list can be found in our [Storage Component API](https://api.keboola.com/?service=storage#get-/v2/storage) in the dedicated [Components section](https://api.keboola.com/?service=storage#get-/v2/storage). +The list of components is managed using the Keboola [Developer Portal](https://components.keboola.com/). + +That being said, any Keboola user can use any component, unless + +- the Keboola user (or their token) has a [limited access to the component](/storage/tokens/). +- the component itself limits where it can run (in what projects and for which users). + +If you have not yet created your component, please go through the [tutorial](/extend/component/tutorial/), which will +navigate you through creating an account in the [Developer Portal](https://components.keboola.com/) and +[initializing the component](/extend/component/tutorial/). + +## Publishing Component +A non-published component can be used without limitations, but it is not offered in the Keboola UI. It can only be used via +the [API](https://api.keboola.com/?service=storage#tag--Component-Configurations) or by directly visiting a link with the +specific component ID: + + https://connection.keboola.com/admin/projects/{PROJECT_ID}/extractors/{COMPONENT_ID} + +This way you can fully test your component before requesting its publication. Also, unpublished +components are not part of our [list of public components](https://components.keboola.com/components). +An existing configuration of a non-public component is accessible the same way as a configuration of any other component. + +**Important:** Changes made in the Developer Portal take up to 5 minutes to propagate to all Keboola instances in all regions. + +Before your component can be published, it must be approved by Keboola. Request the approval from the component list in +the [Developer Portal](https://components.keboola.com/). We will review your component and either publish it or contact you +with required changes. + +![Approval screenshot](/extend/publish/approve.png) + +## Component Review +The goal of the component review is to maintain reasonable end-user experience and component reliability. Before +applying for component registration, make sure the same component does not already exist. If there is a similar one +(e.g., an extractor for the same service), clearly state the differences in the new component's description. During our +component review, the best practices in the next sections are followed. + +### Component Name and Description +Before you name and describe your component, check out our YouTube, Facebook Pages, Dark Sky, and ECB Currency Rates +components for inspiration. + +- Names should not contain words like `extractor`, `application`, and `writer`. +
OK: *Cloudera Impala* +
WRONG: *Cloudera Extractor* +- The short description describes the **service** (helping the user find it) rather than the component. +Obviously for large services like Facebook or Gmail, describe the part of the service relevant to the component. +
OK: *Native analytic database for Apache Hadoop* +
WRONG: *This extractor extracts data from Cloudera Impala* +
OK: *Facebook Pages connect your business with people. Facebook Insights help you get good at it.* +
WRONG: *Facebook connects you with friends, family and other people you know, allows you to share photos and videos, send messages and get updates.* +- The long description provides **additional information about the extracted/written data**: +What will the end user get? What must the end user provide? Is the data going to be imported incrementally? Are there links to +available resources?
Configuration instructions should not be included in the long description, because the long description +is displayed before the end user starts configuring the component. However, if there are any special requirements (external approval, +specific account setting), they should be stated. +
OK: *This component allows you to extract currency exchange rates as published by the European Central Bank (ECB). The +exchange rates are available from a base currency (USD, EUR) to 30 destination currencies (AUD, BGN, BRL, CAD, CNY, +CZK, EUR, GBP, HKD, HRK, HUF, CHF, IDR, ILS, INR, JPY, KRW, MXN, MYR, NOK, NZD, PHP, PLN, RON, RUB, SEK, SGD, THB, TRY, +ZAR). The rates are available for all working days from 4 January 1999 up to present.* +- Component icons must be of representative and reasonable quality. Make sure the icon license allows you to use it. +- Components must correctly state the data flow --- [UI options](/extend/component/ui-options/). Use +`appInfo.dataOut` and `appInfo.dataIn` for this purpose: + - Use `appInfo.dataIn` for extractors, which bring data into a Keboola project (omit `appInfo.dataOut` for extractors). + - Use `appInfo.dataOut` for writers, which send data outside (omit `appInfo.dataIn` for writers). + - Use `appInfo.dataIn` and/or `appInfo.dataOut` for applications. +- Use `appInfo.beta` in [UI options](/extend/component/ui-options/) if you suspect changes to the component behavior. +- Licensing information must be valid, and the vendor description must be current. + +### Component Icon + +- Use a PNG image that is at least 256x256px large and has transparent background. + +### Component Configuration + +- Use only the necessary [UI options](/extend/component/ui-options/) (i.e., if there are no output files, do not use `genericDockerUI-fileOutput`). +- For extractors, always use the [default bucket](/extend/common-interface/folders/#default-bucket) --- do not use the `genericDockerUI-tableOutput` flag. +- Use [encryption](/overview/encryption/) to store sensitive values. No plain-text passwords! +- Use a [configuration schema](/extend/component/ui-options/configuration-schema/). + - List all properties in the `required` field. + - Always use `propertyOrder` to explicitly define the order of the fields in the form. + - Use your short `title` without a colon, period, etc. + - Use `description` to provide an explanatory sentence if needed. +
OK: ![Good Schema](/extend/publish/schema-good.png) +
WRONG: ![Bad Schema](/extend/publish/schema-bad.png) +- Use a configuration description only if the configuration is not trivial/self-explanatory. Provide **links to resources** +(for instance, when creating an Elastic extractor, not everyone is familiar with the ElasticSearch query syntax). The +configuration description supports markdown. Your markdown should not start with a header and should use only level 3 and +level 4 headers (level 2 header is prepended before the configuration description).
OK:
some introduction text

### Input +Description
+description of input tables
+
#### First Table
+some other text
+
WRONG:
## Configuration Description
+some introduction text
+
#### Input Description
+description of input tables +
+ +### Component Internals + +- Make sure that the amount of consumed **memory does not depend** on the amount of processed data. Use streaming or +processing in chunks to maintain a limited amount of consumed memory. If not possible, state the expected usage in +the **Component Limits**. +- The component must distinguish between [user and application errors](/extend/common-interface/environment/#return-values). +- The component must [validate](/extend/common-interface/config-file/#validation) its parameters; an invalid configuration must result in a user error. User error messages must clearly state what's wrong and what the user should do to fix the issue. E.g., `Invalid configuration.` is wrong, `Login failed, check your credentials.` is better. +- The events produced must be reasonable. Provide status messages if possible and with a reasonable frequency. Avoid internal messages with no meaning to the end user. Also avoid flooding the event log or sending data files in the event log. +- Set up [continuous deployment](/extend/component/deployment/) so that you can keep the component up to date. +- Use [semantic versioning](http://semver.org/) to mark and deploy versions of your component. Using other tags (e.g., +`latest`, `master`) in production is not allowed. + +### Checklist + +Before requesting to publish a component, please check all rules using [this checklist](/extend/publish/checklist). diff --git a/src/content/docs/extend/publish/schema-bad.png b/src/content/docs/extend/publish/schema-bad.png new file mode 100644 index 000000000..ed2450754 Binary files /dev/null and b/src/content/docs/extend/publish/schema-bad.png differ diff --git a/src/content/docs/extend/publish/schema-good.png b/src/content/docs/extend/publish/schema-good.png new file mode 100644 index 000000000..8082071b7 Binary files /dev/null and b/src/content/docs/extend/publish/schema-good.png differ diff --git a/src/content/docs/external-integrations/index.md b/src/content/docs/external-integrations/index.md index 2bdf56546..fefcecf3c 100644 --- a/src/content/docs/external-integrations/index.md +++ b/src/content/docs/external-integrations/index.md @@ -5,7 +5,7 @@ slug: 'external-integrations' -Keboola's external integration capabilities let you seamlessly extend the platform with your existing tools and workflows. By connecting to Keboola through our [REST API](https://developers.keboola.com/overview/api) or [MCP (Model Context Protocol)](/ai/mcp-server/) server, you can orchestrate data pipelines, trigger jobs, and embed Keboola into a broader automation ecosystem. +Keboola's external integration capabilities let you seamlessly extend the platform with your existing tools and workflows. By connecting to Keboola through our [REST API](/overview/api/) or [MCP (Model Context Protocol)](/ai/mcp-server/) server, you can orchestrate data pipelines, trigger jobs, and embed Keboola into a broader automation ecosystem. > Keboola integrates with external tools at multiple levels, from low-level API access to ready-made workflow platforms. This flexibility allows you to choose the approach that best fits your team's needs. diff --git a/src/content/docs/external-integrations/n8n/index.md b/src/content/docs/external-integrations/n8n/index.md index c6cecb1af..d852b2a80 100644 --- a/src/content/docs/external-integrations/n8n/index.md +++ b/src/content/docs/external-integrations/n8n/index.md @@ -107,7 +107,7 @@ If you need functionality that isn’t covered by the node’s built-in actions, ## Resources -- [Keboola API Documentation](https://developers.keboola.com/overview/api) +- [Keboola API Documentation](/overview/api/) - [n8n Documentation](https://docs.n8n.io) - [n8n Community Nodes Guide](https://docs.n8n.io/integrations/#community-nodes) - [NPM Package](https://www.npmjs.com/package/@keboola/n8n-nodes-keboola) diff --git a/src/content/docs/flows/flows-legacy/index.md b/src/content/docs/flows/flows-legacy/index.md index 54e76715c..34bacdf1e 100644 --- a/src/content/docs/flows/flows-legacy/index.md +++ b/src/content/docs/flows/flows-legacy/index.md @@ -104,7 +104,7 @@ execute more jobs in parallel. Keboola will then concurrently execute the jobs t - If you are working with APIs that are inconsistent or prone to frequent errors, consider enabling the **Continue on Failure** flag. Each phase (or step) of the flow will only run successfully if all jobs within that phase complete successfully. If a phase fails, no subsequent phases will continue. However, enabling this flag for each task (off by default) allows the flow to continue to subsequent phases, ending with a warning status if errors are encountered. -- Finally, to modify the parameters sent to the underlying [API call](https://developers.keboola.com/integrate/jobs/#run-a-job), you can set **Task Parameters**. +- Finally, to modify the parameters sent to the underlying [API call](/integrate/jobs/#run-a-job), you can set **Task Parameters**. Select the task and click **Set advanced parameters**. When finished, click **Set**. ***Example of the advanced parameter:** changing a variable in transformation:* diff --git a/src/content/docs/flows/index.md b/src/content/docs/flows/index.md index c11a48045..a4d72d557 100644 --- a/src/content/docs/flows/index.md +++ b/src/content/docs/flows/index.md @@ -2,6 +2,7 @@ title: Conditional Flows slug: 'flows' redirect_from: + - /integrate/orchestrator/ - /flows/conditional-flows/ --- @@ -48,7 +49,7 @@ You can also set up parallelization **within a component** (configuration), dire - Failure handling is expressed through [conditions](#conditions) instead of a "Continue on Failure" toggle — you can branch on task or phase status (e.g., `if status == 'error' then ...`) to send notifications, run fallback logic, or end the flow. See also [Retry](#retry) for automatic retries of failed tasks. -- To modify the parameters sent to the underlying [API call](https://developers.keboola.com/integrate/jobs/#run-a-job), you can set **Task Parameters**. +- To modify the parameters sent to the underlying [API call](/integrate/jobs/#run-a-job), you can set **Task Parameters**. Select the task and click **Set advanced parameters**. When finished, click **Set**. ***Example of the advanced parameter:** changing a variable in transformation:* diff --git a/src/content/docs/integrate/artifacts/artifacts-tutorial-1.png b/src/content/docs/integrate/artifacts/artifacts-tutorial-1.png new file mode 100644 index 000000000..b35de9b1a Binary files /dev/null and b/src/content/docs/integrate/artifacts/artifacts-tutorial-1.png differ diff --git a/src/content/docs/integrate/artifacts/artifacts-tutorial-2.png b/src/content/docs/integrate/artifacts/artifacts-tutorial-2.png new file mode 100644 index 000000000..3b78098be Binary files /dev/null and b/src/content/docs/integrate/artifacts/artifacts-tutorial-2.png differ diff --git a/src/content/docs/integrate/artifacts/artifacts-tutorial-3.png b/src/content/docs/integrate/artifacts/artifacts-tutorial-3.png new file mode 100644 index 000000000..d9ba47847 Binary files /dev/null and b/src/content/docs/integrate/artifacts/artifacts-tutorial-3.png differ diff --git a/src/content/docs/integrate/artifacts/artifacts-tutorial-4.png b/src/content/docs/integrate/artifacts/artifacts-tutorial-4.png new file mode 100644 index 000000000..ae120cc96 Binary files /dev/null and b/src/content/docs/integrate/artifacts/artifacts-tutorial-4.png differ diff --git a/src/content/docs/integrate/artifacts/index.md b/src/content/docs/integrate/artifacts/index.md new file mode 100644 index 000000000..ae794729b --- /dev/null +++ b/src/content/docs/integrate/artifacts/index.md @@ -0,0 +1,127 @@ +--- +title: Artifacts +slug: 'integrate/artifacts' +--- + + +*Note: This is a preview feature and as such may change considerably in the future. The project must have an `artifacts` feature enabled.* + +**Artifacts** are additional files that can be produced or consumed by a [component](/extend/component). + +See [Tutorial](/integrate/artifacts/tutorial) for step-by-step example. + +## Introduction +In some cases it's useful if a component not only extracts, transforms or uploads data, but also generate some other output, metadata or other runtime-discovered data. +These could be for example: +- AI models +- performance graphs of such models +- status updates from long-running tasks +- documentation +- data quality checks from in-progress tasks + +These additional information can be stored in artifacts and processed by another component or 3rd party tool. + +## Storage +Artifacts are stored in Keboola File Storage. + +## Types of artifacts +There are three types of artifacts for now `runs`, `custom` and `shared`. +The type specifies which components will have access to the artifact or which artifacts to download for the component to process. +Types are used in a configuration of a consumer component to specify which artifacts to download. + +- **runs** - artifacts from previous runs of the same configuration + +- **custom** - artifacts from previous runs of a different configuration. The configuration which produced the artifacts will be defined in the consumer configuration (configurationId, componentId, branchId) + +- **shared** - artifacts shared within an orchestration + +`runs` and `custom` types are the same from the producer point of view. To produce a `shared` artifact, it has to be written into a `shared` folder. Read more in [File structure](#file-structure) section. + +## File structure +Artifact is a unique set of files associated with a successful job, component and configuration. +A component can either produce or consume artifacts or both. + +### Produce +To produce an artifact, store one or more files in the following `output` directories. Subdirectories are also supported. +- `/data/artifacts/out/current` to create an artifact of type `runs` / `custom`. +- `/data/artifacts/out/shared` to create an artifact of type `shared`, which can be accessed by any component within the same orchestration. + +After the component job is finished all files and directories inside `current` and `shared` folders will be compressed into an archive and uploaded to File Storage with corresponding tags as a `artifact`. + +### Consume +To consume created artifacts you have to specify, in the configuration of a component, which artifacts (type) to download. + - `runs` to download artifacts produced by the same configuration and component. These will be stored in `/data/artifacts/in/runs/jobs/job-%job_id%` directory. + - `custom` to download artifacts produced by another configuration or component. These will be stored in `/data/artifacts/in/custom/jobs/job-%job_id%` directory. + - `shared` to download artifacts created within the same orchestration by any artifact producing component that has already finished. These will be stored in `/data/artifacts/in/shared/jobs/job-%job_id%` directory. + +## Configuration +Each type of artifact has a separate node in configuration. All the types can be used simultaneously. +Each type node has an attribute "enabled", which enables or disables download of the corresponding artifact type. + +### Runs + - **enabled** [true|false] - enable or disable download of this artifact type + - **filter** + - **date_since** - only artifacts from jobs younger than this will be downloaded + - **limit** - maximum number of the latest jobs from which to download artifacts + +### Custom +- **enabled** [true|false] - enable or disable download of this artifact type +- **filter** + - **branch_id**, **component_id**, **config_id** - specify the configuration to download artifacts from + - **date_since** - only artifacts from jobs younger than this will be downloaded + - **limit** - maximum number of the latest jobs from which to download artifacts + +### Shared +- **enabled** [true|false] - enable or disable download of this artifact type + +Full configuration example with all artifact types: + +```json +{ + "parameters": {}, + "artifacts": { + "runs": { + "enabled": true, + "filter": { + "date_since": "-7 days", + "limit": 5 + } + }, + "custom": { + "enabled": true, + "filter": { + "component_id": "keboola.python-transformation", + "config_id": "12345", + "branch_id": "default", + "date_since": "-7 days", + "limit": 5 + } + }, + "orchestration": { + "enabled": true + } + } +} +``` + +## Artifacts life-cycle in a job +Job runner checks if the project has enabled `artifacts` feature. +Job runner checks the configuration of the component. +If artifacts are enabled, it downloads artifacts to corresponding folders as configured (i.e. `runs`, `custom`, `shared`) and unzips them. + +Component process start and the component can: + +- access and process the downloaded artifacts in shared or custom directory + +- write artifacts to `current` or `shared` directory + +Component finishes and job runner does: + +- gzip the content of runs/current + +- tag the gzipped file with jobId, componentId, configId, runId, branchId and other tags if needed + +- upload the file to File Storage + +## File size limit +All the artifacts produced by a job shouldn’t be bigger than 1 GB. diff --git a/src/content/docs/integrate/artifacts/tutorial/index.md b/src/content/docs/integrate/artifacts/tutorial/index.md new file mode 100644 index 000000000..2c46d922d --- /dev/null +++ b/src/content/docs/integrate/artifacts/tutorial/index.md @@ -0,0 +1,211 @@ +--- +title: Artifacts Tutorial +slug: 'integrate/artifacts/tutorial' +--- + + +This tutorial will show you how to work with artifacts. +In the following example we will use Python Transformation component to produce and consume artifacts. +But these principles would work inside any component. + +In the examples, we use the `curl` console tool to interact with our APIs. + +*Note: `artifacts` feature needs to be enabled in your project. Please contact [support@keboola.com](mailto:support@keboola.com) to enable the feature in your project* +*Note 2: `artifacts` configuration can be created or edited only via [Configuration API](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) for now* + +## Examples + +For each example we will need [Storage API Token](/management/project/tokens/) to make the API call. + +1. Obtain a Storage API token from the user interface of your project, see this [Guide](/management/project/tokens). +2. Store the token and url to the environment variable. + + ```shell + export STORAGE_API_HOST="https://connection.keboola.com" + export TOKEN="..." + ``` + +### 1. Produce artifact + +This is very simple example. We will just create a Python Transformation, which will write a file to the artifacts "upload" folder. +This file will be then uploaded as "artifact" to File Storage. + +1. In your Keboola project, create a new Python transformation, and paste this code into it: + ``` + import os + with open("/data/artifacts/out/current/myartifact1", "w") as file: + file.write("this is my artifact file content") + ``` + + ![Artifacts - transformation](/integrate/artifacts/artifacts-tutorial-1.png) + +2. Run the transformation - it should upload the file to File Storage as "artifact" + + ![Artifacts - Job](/integrate/artifacts/artifacts-tutorial-2.png) + +3. The file is now visible in File Storage with appropriate tags + + ![Artifacts - File Storage](/integrate/artifacts/artifacts-tutorial-2.png) + +### 2. Produce & consume artifacts + +To consume (download) artifacts for component to work with, we need to enable and configure artifacts download in the configuration of a component. + +We will create another configuration of the Python transformation via API. + +The artifacts part of the configuration will look like this. +It will enable download of artifacts of type `runs` with limit 5, which means this will download artifacts created by the last 5 runs of the same component configuration + + ```json + { + "artifacts":{ + "runs":{ + "enabled":true, + "filter":{ + "limit":5 + } + } + } + } + ``` + +The script of the transformation will look like following. +Files read from `/data/artifacts/in/runs/*/*` will be displayed at output - these are the artifact files downloaded. +The script will also generate a new artifact and write it to `/data/artifacts/out/current/myartifact1` as in previous example. + + ```python + import os + import glob + + # Download + print(glob.glob("/data/artifacts/in/runs/*/*")) + + # Upload + with open("/data/artifacts/out/current/myartifact1", "w") as file: + file.write("value1") + ``` +1. Run this curl command to create the configuration: + + ```shell + curl -X POST "$STORAGE_API_HOST/v2/storage/branch/default/components/keboola.python-transformation-v2/configs" \ + -H "X-StorageApi-Token: $TOKEN" \ + -H 'Content-Type: application/x-www-form-urlencoded' \ + --data-urlencode 'configuration={"parameters":{"blocks":[{"name":"Block 1","codes":[{"name":"artifacts","script":["import os\nimport glob\n\n# Download\nprint(glob.glob(\"/data/artifacts/in/runs/*/*\")) \n\n# Upload\nwith open(\"/data/artifacts/out/current/myartifact1\", \"w\") as file:\n file.write(\"value1\")"]}]}]},"artifacts":{"runs":{"enabled":true,"filter":{"limit":5}}}}' \ + --data-urlencode 'name=Artifacts upload & download' \ + --data-urlencode 'description=Test Artifacts upload & download' + ``` + +### 3. Consume artifacts from different component +Similar to previous example we will create a configuration of Python Transformation component. +But this time we will download artifacts produced by the configuration from `Example 2`. + +1. Export the id of the previously created configuration into an environment variable: + ```shell + export CONFIG_ID="..." + ``` + +2. Run curl command + ```shell + curl -X POST "$STORAGE_API_HOST/v2/storage/branch/default/components/keboola.python-transformation-v2/configs" \ + -H "X-StorageApi-Token: $TOKEN" \ + -H 'Content-Type: application/x-www-form-urlencoded' \ + --data-urlencode 'configuration={"parameters":{"blocks":[{"name":"Block 1","codes":[{"name":"artifacts","script":["import os\nimport glob\n\n# Download\nprint(glob.glob(\"/data/artifacts/in/custom/*/*\"))"]}]}]},"artifacts":{"custom":{"enabled":true,"component_id":"keboola.python-transformation","config_id":"$CONFIG_ID","branch_id":"default","filter":{"limit":5}}}}' \ + --data-urlencode 'name=Artifacts upload & download' \ + --data-urlencode 'description=Test Artifacts upload & download' + ``` + +The whole configuration now looks like this: + + ```json + { + "parameters": { + "blocks": [ + { + "name": "Block 1", + "codes": [ + { + "name": "artifacts", + "script": [ + "import os\nimport glob\n\n# Download\nprint(glob.glob(\"/data/artifacts/in/custom/*/*\"))" + ] + } + ] + } + ] + }, + "artifacts": { + "custom": { + "enabled": true, + "component_id": "keboola.python-transformation", + "config_id": "$CONFIG_ID", + "branch_id": "default", + "filter": { + "limit": 5 + } + } + } + } + ``` + +### 4. Shared artifacts +This example will show how to share artifacts within an orchestration +We will create two configurations of Python Transformation component. +One will produce a shared artifact and the other will consume it. +Both configurations needs to be in the same orchestration. +The configuration producing artifact needs to be in a phase that precedes the consuming one. + +1. Create "Producer" configuration + The Python code will write a file into a shared folder: + + ```python + import os + with open(path+\"/myartifact-shared\", \"w\") as file: + file.write(\"value1\")" + ``` + + Run curl command to create the configuration: + + ```shell + curl -X POST "$STORAGE_API_HOST/v2/storage/branch/default/components/keboola.python-transformation-v2/configs" \ + -H "X-StorageApi-Token: $TOKEN" \ + -H 'Content-Type: application/x-www-form-urlencoded' \ + --data-urlencode 'configuration={"parameters":{"blocks":[{"name":"Block 1","codes":[{"name":"Upload shared","script":["import os\npath = \"/data/artifacts/out/shared\"\nwith open(path+\"/myartifact3\", \"w\") as file:\n file.write(\"value1\")"]}]}]},"artifacts":{"runs":{"enabled":true,"filter":{"limit":5}}}}' \ + --data-urlencode 'name=Artifacts shared Producer' \ + --data-urlencode 'description=Artifacts upload shared' + ``` + +2. Create "Consumer" configuration + + The artifacts configuration: + ```json + { + "artifacts": { + "shared": { + "enabled": true + } + } + } + ``` + + The Python script: + + ```python + import os + import glob + print(glob.glob("/data/artifacts/in/shared/*/*")) + ``` + + Run curl command to create the configurtion: + + ```shell + curl -X POST "$STORAGE_API_HOST/v2/storage/branch/default/components/keboola.python-transformation-v2/configs" \ + -H "X-StorageApi-Token: $TOKEN" \ + -H 'Content-Type: application/x-www-form-urlencoded' \ + --data-urlencode 'configuration={"parameters":{"blocks":[{"name":"Block 1","codes":[{"name":"Download shared","script":["import os\nimport glob\n\nprint(glob.glob(\"/data/artifacts/in/shared/*/*\")) "]}]}]},"artifacts":{"shared":{"enabled":true}}}' \ + --data-urlencode 'name=Artifacts shared Consumer' \ + --data-urlencode 'description=Artifacts download shared' + ``` + +3. Now put each of the configurations into an Orchestration. "Artifacts shared Producer" into phase 1 and "Artifacts shared Consumer" into phase 2. + + ![Artifacts orchestration](/integrate/artifacts/artifacts-tutorial-4.png) diff --git a/src/content/docs/integrate/data-streams/push_data.drawio.png b/src/content/docs/integrate/data-streams/push_data.drawio.png new file mode 100644 index 000000000..19abce2f6 Binary files /dev/null and b/src/content/docs/integrate/data-streams/push_data.drawio.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-add.png b/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-add.png new file mode 100644 index 000000000..d98d32b94 Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-add.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-individual-events.png b/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-individual-events.png new file mode 100644 index 000000000..43908f2d6 Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-individual-events.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-issues.png b/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-issues.png new file mode 100644 index 000000000..0ab61f8f9 Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook-issues.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook.png b/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook.png new file mode 100644 index 000000000..4b9fecec2 Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/gh-settings-webhook.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/gh-tabs.png b/src/content/docs/integrate/data-streams/tutorial/gh-tabs.png new file mode 100644 index 000000000..a78cb9d47 Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/gh-tabs.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_file.png b/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_file.png new file mode 100644 index 000000000..0440b7d7d Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_file.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_table.png b/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_table.png new file mode 100644 index 000000000..c13dac3a4 Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_table.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_table_data.png b/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_table_data.png new file mode 100644 index 000000000..44192eb0a Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_table_data.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_token.png b/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_token.png new file mode 100644 index 000000000..b5ecdce29 Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/github_webhook_export_token.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/table.png b/src/content/docs/integrate/data-streams/tutorial/table.png new file mode 100644 index 000000000..7e39f0b9b Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/table.png differ diff --git a/src/content/docs/integrate/data-streams/tutorial/token.png b/src/content/docs/integrate/data-streams/tutorial/token.png new file mode 100644 index 000000000..d6b94c7f2 Binary files /dev/null and b/src/content/docs/integrate/data-streams/tutorial/token.png differ diff --git a/src/content/docs/integrate/database/ssh-tunnel.jpg b/src/content/docs/integrate/database/ssh-tunnel.jpg new file mode 100644 index 000000000..d9d9d38a0 Binary files /dev/null and b/src/content/docs/integrate/database/ssh-tunnel.jpg differ diff --git a/src/content/docs/integrate/index.md b/src/content/docs/integrate/index.md new file mode 100644 index 000000000..825535809 --- /dev/null +++ b/src/content/docs/integrate/index.md @@ -0,0 +1,33 @@ +--- +title: Integration +slug: 'integrate' +--- + +You can look at Keboola as a system of independent and loosely coupled microservices (components). + +Each microservice has its own code base, and a publicly accessible API and configuration. +We do not cheat or have any advantage over other developers; our UI and other components use only these public APIs. + +As a result, it is very easy to, for example, write custom scripts to bootstrap a project, or do something that our UI does not offer. +Let's have a look into this! + +One of the very important components is [Storage](/integrate/storage/), which not only stores all data in a +project, but also provides additional functions such as managing other components and their configurations. +When you are integrating your systems with Keboola, **chances are that you want to start with [Storage](/integrate/storage/)**. + + \ No newline at end of file diff --git a/src/content/docs/integrate/jobs/index.md b/src/content/docs/integrate/jobs/index.md new file mode 100644 index 000000000..c8ec907d3 --- /dev/null +++ b/src/content/docs/integrate/jobs/index.md @@ -0,0 +1,496 @@ +--- +title: Component Jobs +slug: 'integrate/jobs' +redirect_from: + - /overview/jobs/ +--- + + +Most operations, such as extracting data or running an application are executed in Keboola as +background, asynchronous [jobs](/management/jobs/). When an operation is triggered, for example, you run an extractor, a +*job* is created. The job starts executing or waits in the queue until it can start executing. +The job execution and queuing are fully automatic. The job execution is asynchronous, so you need to + +- *create* (run) a job, and +- *wait* for it to finish. + +The core API for working with jobs is the [Queue API](https://api.keboola.com/?service=job-queue#job-queue). It provides operations for +running/creating, terminating and listing jobs. +[Components](/overview/) differ in their upper limits on how long a job can be executing and how much memory it is allowed to consume. +These limits are set by the component developer and act primarily as a safeguard. + +## Job Properties +When you create a job it automatically transitions through states until it reaches some of the final states. +When you create or retrieve a job, you'll obtain a JSON with Job object, whose properties are described below in more detail. + +
+ Click to expand the response. + +```json +{ + "id": "10440535", + "runId": "10440530.10440533.10440534.10440535", + "parentRunId": "10440530.10440533.10440534", + "project": { + "id": "66", + "name": "Sandbox" + }, + "token": { + "id": "7455", + "description": "[_internal] Scheduler" + }, + "status": "success", + "desiredStatus": "processing", + "mode": "run", + "component": "keboola.ex-db-snowflake", + "config": "493493", + "configData": [], + "configRowIds": [ + "41510" + ], + "tag": "5.5.0", + "createdTime": "2022-01-24T22:41:10+00:00", + "startTime": "2022-01-24T22:41:13+00:00", + "endTime": "2022-01-24T22:41:48+00:00", + "durationSeconds": 35, + "result": { + "input": { + "tables": [] + }, + "images": [ + [ + { + "id": "developer-portal-v2/keboola.ex-db-snowflake:5.5.0", + "digests": [ + "developer-portal-v2/keboola.ex-db-snowflake@sha256:0f9428c52afea457ec3865cab7cfe457f4f875f3cf45d36f1876c709211da9cf" + ] + } + ] + ], + "output": { + "tables": [ + { + "id": "in.c-keboola-ex-db-snowflake-493493.opportunity", + "name": "opportunity", + "columns": [ + { + "name": "Id" + }, + { + "name": "Name" + }, + { + "name": "AccountId" + }, + { + "name": "OwnerId" + }, + { + "name": "Amount" + }, + { + "name": "StageName" + }, + { + "name": "CreatedDate" + }, + { + "name": "CloseDate" + }, + { + "name": "Probability" + }, + { + "name": "Start_Date" + }, + { + "name": "End_Date" + }, + { + "name": "Record_Type_Name" + }, + { + "name": "AdvertiserName" + }, + { + "name": "Advertiser_Vertical" + }, + { + "name": "Type" + } + ], + "displayName": "opportunity" + } + ] + }, + "message": "Component processing finished.", + "configVersion": "16" + }, + "usageData": [], + "isFinished": true, + "url": "https://queue.north-europe.azure.keboola.com/jobs/10440535", + "branchId": null, + "variableValuesId": null, + "variableValuesData": { + "values": [] + }, + "backend": [], + "metrics": { + "backend": { + "size": null + }, + "storage": { + "inputTablesBytesSum": 0 + } + }, + "behavior": { + "onError": null + }, + "parallelism": "2", + "type": "standard" +} +``` +
+ +### Job Status +A job can have different values for `status`: +- `created` (the job is created, but has not started executing yet) +- `waiting` (the job is waiting for other jobs to finish) +- `processing` (job stuff is being done) +- `success` (the job is finished) +- `error` (the job is finished) +- `warning` (the job is finished, but one of its child jobs failed) +- `terminating` (the user has requested to abort the job) +- `cancelled` (the job was created, but it was aborted before its execution actually began) +- `terminated` (the job was created and it was aborted in the middle of its execution) + +![Job State transitions](/integrate/jobs/states.png) + +When you create a job it is in the `created` state. In a success scenario it will transition to a `processing` state and when the actual work is done, to the + `success` state. If you change your mind and terminate a job, it will enter `terminating` state and then ends with either `terminated` (execution terminated) + or `cancelled` (execution did not actually start). The difference is that you can be sure that a `cancelled` job did absolutely no operations, + whereas a terminated job, could've done even all of the work it was supposed to do. + + If a job cannot be executed, it will enter the `waiting` state. The waiting state means that the job cannot be executed due to reasons on the Keboola + project side. This means that the reasons for waiting jobs lie solely in what jobs are already running in the given project. There are three core reasons for waiting jobs: + - If you run two jobs of the same configuration, the second one will wait until the first one is finished. This behavior is + called "configuration lock" and protects your project from [race conditions](https://en.wikipedia.org/wiki/Race_condition). + - Orchestration [phases](/orchestrator/tasks/#organize-tasks). When you run an orchestration, the jobs for all phases are created. Phases that depend on other phases enter the `waiting` state. + - Setting parallel limits. If you run a configuration with 10 tables and set parallelism to 2, then 10 jobs will be created, 2 will enter `processing` + state and 8 of them will immediately enter the `waiting` state. + + If a job cannot be run due to platform reasons (e.g. insufficient resources, platform outage), it will remain in the `created` state. In rare situations + (e.g. hardware failure), the job may return back to created state. Moving the job out of the `created` state is out of the control of the end-user. + + Of all the states a job can be in, only the state `processing` is considered to be job runtime (see `durationSeconds` field) and therefore billable. + That means `waiting` or `created` jobs do not have any costs associated with them, they represent a plan of what is going to happen. + + The states `terminated`, `cancelled`, `success` and `error` are final and end the job transitions. When a job is in final state, the `isFinished` flag is true. + The job object is both immutable and eventually consistent. Once you create a job, you cannot change any of it's properties. Any changing properties are + self-modifying and they will stop modifying once the job reaches one of the final states. + +Apart from the `status` field, the job also has `desiredStatus` field. This is either `processing` or `terminating`. The desired status is +processing until a job termination is requested. This changes the desired status to `terminating`. Other changes are not permitted. + +### Job ID +When a job is created, an `id` and `runId` and optionally `parentRunId` are assigned to it. The `runId` and `parentRunId` represent +parent-child relationship between jobs. Parent-child hierarchical relationship can be defined via the `X-KBC-RunId` header, when used the +newly created job will become child of the job with the provided `RunId`. + +The `runId` field contains job `id`s with representing the job hierarchy. If there is no hierarchy, then `runId` is equal to `id`. If there is +hierarchy then `runId` is `parentId` concatenated with `id`. The hierarchy delimiter is dot `.`. Examples: + +- `id=123`, `runId=123`, `parentRunId=null` -- Job has no parent +- `id=345`, `runId=123.345`, `parentRunId=123` -- Job is a child of job `123`. +- `id=678`, `runId=123.345.678`, `parentRunId=123.345` -- Job is a child of job `345` which in turn is a child of job `123`. + +Jobs may be nested without limits. The parent-child relationship itself is a weak relationship. By itself it does not mean anything +special outside of UI grouping and the function that terminating a parent job issues a termination request to all its children. +Running a job as a child of another job does not by itself cause the parent to wait for child +completion or any other added functionality. +Such functionality is implemented in specific components (e.g. Orchestrator) or for specific [job types](todo). + +### Job Configuration +To create a job, you must provide the [configuration](/components/) to run. A configuration is always tied to a specific +[component](/extend/component/). + +A configuration can be provided in multiple ways. The easiest is to provide a reference to +a stored configuration ID using the `config` field as shown above. Configurations can be stored and listed using the +[Component Configurations API endpoint](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs). +When using a configuration which contains [Configuration Rows](/components/#configuration-rows), the job can optionally execute +only certain rows. Use the `configRowIds` field to list row IDs to execute. Note that if you do not list any rows, then all rows will be executed except +for disabled rows. When you enumerate rows to execute, then the enumerated rows will be executed even if they are disabled. To run a job of +a configuration in a branch, provide the [branch ID](https://api.keboola.com/?service=storage#get-/v2/storage/dev-branches) +in the `branchId` field. If you do not provide `branchId`, then the default branch is used. +Take care that only the **combination of component ID, configuration ID and branch ID is unique**. It is possible for two configurations with the +same ID to exist (either for different component or for a different branch). + +Another option is to provide the entire configuration in the `configData` field. In that case the whole configuration data +has to be provided in the request. If you are retrieving a +[stored configuration](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-), take +note that the configuration data is the contents of the `configuration` node and not the entire +response. When using the `configData` field, the `configRowIds` and `branchId` values are ignored. When using the `configData` field the `config` field +is ignored for the purpose of reading the configuration, but may still be required in case the component is using +[Default Bucket](/extend/component/tutorial/output-mapping/#configuring-default-bucket). In that case, the +configuration referenced in `config` is used to generate the name of the output bucket. It still holds that configuration data is not read from it. +That means that `configData` always fully overrides the `config` field. + +### Job Mode +When creating a job, you need to provide `mode`. This can be one of `run`, `forceRun` and `debug`. The basic `mode` choice is `run`. +Use the `forceRun` mode to run a configuration that is disabled. The `debug` can be used during [Component Development & Debugging](/extend/component/tutorial/debugging/). + +### Job Runtime configuration +You may provide runtime settings for a job. Runtime settings do not affect what the job does, they affect how the job does it. The available runtime settings are: + +- `backend.type` --- for Snowflake transformations this is the size of the [Snowflake warehouse](/transformations/snowflake-plain/#dynamic-backends) used for the job; otherwise it affects the [container size](/transformations/python-plain/#dynamic-backends). Available values for backend type are `small`, `medium`, `large`. +- `parallelism` --- runs [Configuration Rows](/components/#configuration-rows) (if present in the configuration) in parallel. Allowed values are integer values and `infinity`, which runs all rows in parallel. When not specified, the rows are run sequentially. +- `tag` --- runs the component with a specific version of code. This is mostly used during component development, testing and debugging. + +Runtime parameters can be specified on various levels. The values can be specified in the component configuration. They can also be specified +when creating a job, in which case it overrides the configuration. It may also be specified for an orchestration, in which case it overrides what is specified +in individual jobs of that orchestration. + +When stored in the component configuration, the runtime settings live in a top-level `runtime` node of the configuration JSON --- a sibling of `parameters`, +not inside it. For example, to pin all jobs of a configuration to a specific image tag: + +```json +{ + "parameters": { + "...": "..." + }, + "runtime": { + "tag": "my-branch-3" + } +} +``` + +Jobs of this configuration then run the given image tag (the job detail shows the resolved value in its `tag` field) until the `runtime.tag` key +is removed from the configuration. This is the usual way of testing a development build of a component in a single project without affecting +other projects --- the tag in the [Developer Portal](/extend/publish/) stays untouched. Do not forget to remove the key when done; a pinned +configuration keeps running the old image even after new versions of the component are released. + +### Job Type +Job can be of one of the four types `standard`, `container`, `phaseContainer` and `orchestrationContainer`. The `standard` is something which does actual work. +Only standard jobs consume billable time and are counted towards consumption of any resources. Other job types are virtual containers encapsulating standard jobs. + +The `container` job represent a job containing [parallel executions](/integrate/jobs/#job-runtime-configuration) +of configuration rows. `phaseContainer` type contains standard jobs in a single +phase of an orchestration. `orchestrationContainer` job type represents an [orchestration](/orchestrator/) and +contains phase jobs of that orchestration. What these job types have in common is a strong +[parent-child relationship](/integrate/jobs/#job-id). This means for example that when a child job fails, the container fails too. The +behavior can be further controlled by the `onError` setting. You cannot specify job type when creating a job, it is selected automatically as needed. + +## Working with the Jobs API +The main API to run the jobs is [Job Queue API](https://api.keboola.com/?service=job-queue#job-queue). There are some API calls from other services +which might be useful when working with jobs: + +- [Create configurations](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) +- [List Job Events](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/events) +- [Encrypt values](https://api.keboola.com/?service=encryption#post-/encrypt) +- [Run Synchronous Actions](https://api.keboola.com/?service=sync-actions#sync-actions/POST/actions) +- [Subscribe to Job Events](https://api.keboola.com/?service=notification#notification/tag/project-subscriptions/POST/project-subscriptions) +- [Schedule jobs](https://api.keboola.com/?service=scheduler#scheduler/tag/schedules/POST/schedules) + +The component jobs are asynchronous operations, this means that you create it and then you have to actively wait for the result. Note that there +are other *unrelated* cases of asynchronous operations in Keboola Platform which are in principle the same, but may differ in little details. +The most common one is: +[Storage Jobs](https://api.keboola.com/?service=storage#get-/v2/storage/jobs/-jobId-), triggered, for instance, by +[asynchronous imports](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/buckets/-id-/tables-async) +or [exports](https://keboola.docs.apiary.io/#reference/tables/unload-data-asynchronously/asynchronous-export) + +### Run a Job +You need to know the *component Id* and *configuration Id* to create a job. You can get these from the UI links. To use the API to obtain a +list of all components available in the project, and their configuration, you can use the +[Get components](https://api.keboola.com/?service=storage#get-/v2/storage). +See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb). +A snippet of the response is below: + +```json +[ + { + "id": "keboola.ex-db-snowflake", + "type": "extractor", + "name": "Snowflake", + "description": "Cloud-Native Elastic Data Warehouse Service", + "documentationUrl": "https://github.com/keboola/db-extractor-snowflake/blob/master/README.md", + "configurations": [ + { + "id": "554424643", + "name": "Sample database", + "description": "", + "created": "2019-12-03T11:18:28+0100", + "creatorToken": { + "id": 199182, + "description": "ondrej.popelka@keboola.com" + }, + "version": 3, + "changeDescription": "Quickstart config creation", + "isDisabled": false, + "isDeleted": false, + "currentVersion": { + "created": "2019-12-03T11:19:50+0100", + "creatorToken": { + "id": 199182, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Quickstart config creation" + } + } + ] + } +] +``` + +From there, the important part is the `id` field and `configurations.id` field. For instance, in the +above, there is a database extractor with the `id` `keboola.ex-db-snowflake` and a +configuration with the id `554424643`. + +Then use the [create a job](https://api.keboola.com/?service=job-queue#job-queue/tag/jobs/POST/jobs) +API call and pass the configuration ID and component ID in request body: + +```json +{ + "component": "keboola.ex-db-snowflake", + "config": "554424643", + "mode": "run" +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb). +When a job is created, you will obtain a response similar to this: + +```json +{ + "id": "807932655", + "runId": "807932655", + "parentRunId": "", + "project": { + "id": "7150", + "name": "Sandbox" + }, + "token": { + "id": "199182", + "description": "ondrej.popelka@keboola.com" + }, + "status": "created", + "desiredStatus": "processing", + "mode": "run", + "component": "keboola.ex-db-snowflake", + "config": "554424643", + "configData": [], + "configRowIds": [], + "tag": "5.5.0", + "createdTime": "2022-01-25T16:34:40+00:00", + "startTime": null, + "endTime": null, + "durationSeconds": 0, + "result": [], + "usageData": [], + "isFinished": false, + "url": "https://queue.keboola.com/jobs/807932655", + "branchId": null, + "variableValuesId": null, + "variableValuesData": { + "values": [] + }, + "backend": [], + "metrics": [], + "behavior": { + "onError": null + }, + "parallelism": null, + "type": "standard" +} +``` + +This means that the job was `created` and will automatically start executing. +From the above response, the most important part is `url`, which gives you the URL of the resource for +[Job status polling](https://en.wikipedia.org/wiki/Polling_(computer_science)). + +### Job Polling +If you want to get the actual job result, poll the [Job API](https://api.keboola.com/?service=job-queue#job-queue/tag/jobs/GET/jobs/{jobId}) +for the current state of the job. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb). + +You will receive a response in the same format as when you crated the job: + +```json +{ + "id": "807933826", + "runId": "807933826", + "parentRunId": "", + "project": { + "id": "7150", + "name": "7150" + }, + "token": { + "id": "199182", + "description": "ondrej.popelka@keboola.com" + }, + "status": "processing", + "desiredStatus": "processing", + "mode": "run", + "component": "keboola.ex-db-snowflake", + "config": "554424643", + "configData": [], + "configRowIds": [], + "tag": "5.5.0", + "createdTime": "2022-01-25T16:41:12+00:00", + "startTime": "2022-01-25T16:41:22+00:00", + "endTime": null, + "durationSeconds": 0, + "result": [], + "usageData": [], + "isFinished": false, + "url": "https://queue.keboola.com/jobs/807933826", + "branchId": null, + "variableValuesId": null, + "variableValuesData": { + "values": [] + }, + "backend": [], + "metrics": [], + "behavior": { + "onError": null + }, + "parallelism": null, + "type": "standard" +} +``` + +From the above response, the most important part is the `status` field (`processing`, in this case). +To obtain the Job result, periodically send the above API call until the job status changes +to one of the finished states or until `isFinished` is true. + +### Run a Debug job +To run a debug job, use `debug` for the mode. Optionally you can provide the component version which should run +to [live test](/extend/component/deployment/#test-live-configurations) an image. + +```json +{ + "component": "keboola.ex-db-snowflake", + "config": "554424643", + "mode": "debug", + "tag": "5.5.0" +} +``` + +The debug mode creates a job that prepares the data folder including the serialized configuration files. Then it compresses the +[data folder](/extend/component/running/#preparing-the-data-folder) and uploads it to your project's Files in Storage. This way you will get a snapshot +of what the data folder looked like before the component started. If processors are used, a snapshot of the data folder is created before each processor. After the entire component finishes, another snapshot is made. For example, if you run component A with processor B and C in the after section, you will receive: + +- `stage_0` file with contents of the data folder before component A was run +- `stage_1` file with contents of the data folder before processor B was run +- `stage_2` file with contents of the data folder before processor C was run +- `stage_output` file with contents of the data folder before output mapping was about to be performed (after C finished). + +If configuration rows are used, then the above is repeated for each configuration row. If the job finishes with and error, only the stages before the error are uploaded. + +This API call does not upload any tables or files to Storage. I.e. when the component finishes, its output is discarded and the output mapping to storage +is not performed. This makes this API call generally very safe to call, because it cannot break the Keboola project in any way. However, keep +in mind, that if the component has any outside side effects, these will get executed. This applies typically to writers which will write the data +into the external system even when running in debug mode. + +Note that the snapshot archive will contain all files in the data folder including any temporary files produced be the component. The snapshot will not +contain the output state.json file. This is because the snapshot is made before a component is run where the out state of the previous component is +not available any more. Also note that all encrypted values are removed from the configuration file and there is no way to retrieve them. It is +also advisable to run this command with limited input mapping so that you don't end up with gigabyte size archives. diff --git a/src/content/docs/integrate/jobs/states.png b/src/content/docs/integrate/jobs/states.png new file mode 100644 index 000000000..b3e0e6e95 Binary files /dev/null and b/src/content/docs/integrate/jobs/states.png differ diff --git a/src/content/docs/integrate/storage/api/async-import-handling.svg b/src/content/docs/integrate/storage/api/async-import-handling.svg new file mode 100644 index 000000000..777e93372 --- /dev/null +++ b/src/content/docs/integrate/storage/api/async-import-handling.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/src/content/docs/integrate/storage/api/configurations/index.md b/src/content/docs/integrate/storage/api/configurations/index.md new file mode 100644 index 000000000..a010cf161 --- /dev/null +++ b/src/content/docs/integrate/storage/api/configurations/index.md @@ -0,0 +1,523 @@ +--- +title: Component Configurations API +slug: 'integrate/storage/api/configurations' +--- + + +[Configurations](/storage/configurations/) are an important part of a Keboola project. Most operations are +available in the UI. Use the API if you want to manipulate the configurations programmatically. + +Configurations represent component **instances** in a project. Each Keboola component has different configuration +options and requirements, which must be respected. As such, Keboola configurations provide a general framework for configuring +components, while the specific implementation details are left to the components themselves. + +When working with the [Component Configurations API](https://api.keboola.com/?service=storage#tag--Component-Configurations), +you need to know the `componentId` of the component being configured. +You can see a list of public components in [the Developer Portal](https://components.keboola.com/components), or you can get +a list of all available components with the [API index call](https://api.keboola.com/?service=storage#get-/v2/storage). +See our [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb). + +It will give you something like this: + +```json +{ + "host": "4edece0b0052", + "api": "storage", + "version": "v2", + "revision": "21fb56a0f6d61a307f350247a45950b1e4049625", + "documentation": "https://connection.keboola.com/api/storage/doc.json", + "components": [ + { + "id": "keboola.ex-aws-s3", + "type": "extractor", + "name": "AWS S3", + "description": "AWS Simple Storage Service", + "longDescription": "Download ... from AWS S3 and upload them to Storage.", + "version": 23, + "hasUI": false, + "hasRun": false, + "ico32": "https://ui.keboola-assets.com/.../keboola.ex-aws-s3/32/20.png", + "ico64": "https://ui.keboola-assets.com/.../keboola.ex-aws-s3/64/20.png", + "data": { + "definition": { + "type": "aws-ecr", + "uri": "147946154733.../keboola.ex-aws-s3", + "tag": "v3.0.0", + "repository": { + "region": "us-east-1" + } + }, + "vendor": { + "contact": [ + "Keboola", + "Křižíkova 488/115\n186 00 Prague 8\nCzech Republic", + "support@keboola.com" + ], + "licenseUrl": "https://github.com/keboola/aws-s3-extractor/blob/master/LICENSE" + }, + "configuration_format": "json", + "network": "bridge", + "memory": "512m", + "forward_token": false, + "forward_token_details": false, + "default_bucket": true, + "default_bucket_stage": "in", + "staging_storage": { + "input": "local" + } + }, + "flags": [ + "genericDockerUI", + "genericDockerUI-processors", + "appInfo.dataIn" + ], + "configurationSchema": {}, + "emptyConfiguration": {}, + "uiOptions": {}, + "configurationDescription": null, + "documentationUrl": "https://help.keboola.com/extractors/other/aws-s3/" + } + ], + "services": [...], + "urlTemplates": {...} +} +``` + +From here, you can see all available information about a particular component. In the following examples, we +will use `keboola.ex-aws-s3` --- the AWS S3 extractor. + +## Configuration Structure +Component configurations are largely dependent on the actual component being configured. This makes creating configurations manually +a bit tricky. Rather than starting from scratch, we recommend creating a configuration through the UI and then modifying it when you understand it. + +### Inspecting Configuration +To obtain an existing configuration, you can either use the list of configurations above or +the [Configuration Detail](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-) +API call. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) for obtaining a +configuration of the `keboola.ex-aws-s3` component. You will receive a response similar to this: + +```json +{ + "id": "364479526", + "name": "test", + "description": "", + "created": "2018-03-08T14:54:19+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "version": 5, + "changeDescription": "Table first table edited", + "isDeleted": false, + "configuration": { + "parameters": { + "accessKeyId": "AKIAIBZYEEXQILP46FCA", + "#secretAccessKey": "KBC::ComponentProjectEncrypted==p5gvUw4RSGiVJjT2ayVORpqS7yiKhExi7NnQECntVm8haHaHtFNVDMT8X8b+htnixpXhPIQ9yV+ETrvr+hNeYfh+Ex+UpC//QPWnLcEOC8XOLgmQN8BNgRGSERWUziK0" + } + }, + "rowsSortOrder": [], + "rows": [ + { + "id": "364481153", + "name": "first table", + "description": "", + "configuration": { + "parameters": { + "bucket": "travis-php-db-import-tests-s3filesbucket-vm9zhtm5jd7s", + "key": "tw_accounts.csv", + "saveAs": "first-table", + "includeSubfolders": false, + "newFilesOnly": true + }, + "processors": { + "after": [ + { + "definition": { + "component": "keboola.processor-move-files" + }, + "parameters": { + "direction": "tables", + "addCsvSuffix": true + } + }, + { + "definition": { + "component": "keboola.processor-create-manifest" + }, + "parameters": { + "delimiter": ",", + "enclosure": "\"", + "incremental": false, + "primary_key": [], + "columns": [], + "columns_from": "header" + } + }, + { + "definition": { + "component": "keboola.processor-skip-lines" + }, + "parameters": { + "lines": 1 + } + } + ] + } + }, + "isDisabled": false, + "version": 3, + "created": "2018-03-08T14:58:33+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited", + "state": { + "lastDownloadedFileTimestamp": "1511176959", + "processedFilesInLastTimestampSecond": [ + "tw_accounts.csv" + ] + } + } + ], + "state": {}, + "currentVersion": { + "created": "2018-03-08T23:27:37+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited" + } +} +``` + +The actual component configuration is split into three parts: + +- `configuration` node, containing an arbitrary component configuration +- `state` node, containing a component [state file](/extend/common-interface/config-file/#state-file) +- `rows` node, containing iterations of `configuration` and `state` + +The important part is the ID of the configuration you want to work with. In the following examples, we will use +`364479526`. + +### Configuration +The `configuration` node maps to the [configuration file](/extend/common-interface/config-file/#configuration-file-structure). +It can contain the `storage`, `parameters`, `processors` and `authorization` child nodes (the `image_parameters` and `action` nodes found in the config file +are injected at runtime and are not stored in the configuration). The `authorization` node is set in the configuration only when +[credentials injection](/extend/common-interface/oauth/#credentials-injection) should be used, otherwise it is also set during the runtime. +The `processors` node defines the [processors and their configuration](/extend/component/processors/). +The most common sub-nodes stored in the `configuration` node are therefore `parameters` (containing an arbitrary component configuration) +and `storage` (containing [input](/extend/component/tutorial/input-mapping/) and [output mapping](/extend/component/tutorial/output-mapping/)). +Both are transferred to the +configuration file without modification; that means that the [`storage` configuration](/extend/common-interface/config-file/#configuration-file-structure) +is directly usable in the `configuration` node. The `parameters` node is fully dependent on the component and has no universal specification or rules. + +In the above example, the `configuration` node contains the following: + +```json +"parameters": { + "accessKeyId": "AKIAIBZYEEXQILP46FCA", + "#secretAccessKey": "KBC::ComponentProjectEncrypted==p5gvUw4RSGiVJjT2ayVORpqS7yiKhExi7NnQECntVm8haHaHtFNVDMT8X8b+htnixpXhPIQ9yV+ETrvr+hNeYfh+Ex+UpC//QPWnLcEOC8XOLgmQN8BNgRGSERWUziK0" +} +``` + +That means that the component is not using input mapping nor output mapping. The allowed contents of `parameters` are described +in the [AWS S3 extractor code documentation](https://github.com/keboola/aws-s3-extractor#configuration-options). + +### Configuration Rows +The `rows` node contains iterations of the configuration. The interpretation of configuration rows is again dependent on the +component implementation. In the presented case of the `keboola.ex-aws-s3` component, each row corresponds to a single extracted table. +When `rows` node is non-empty, the component behavior is slightly modified. It behaves as if it were executed as many times as +there are rows. For each row, the `configuration` node from `root` and the `configuration` node from `rows` are merged, with +the latter overwriting the former in the case of conflict. + +Given the above configuration, the **effective configuration** passed to the component +[configuration file](/extend/common-interface/config-file/#configuration-file-structure) will be as follows: + +```json +{ + "parameters": { + "accessKeyId": "AKIAIBZYEEXQILP46FCA", + "#secretAccessKey": "KBC::ComponentProjectEncrypted==p5gvUw4RSGiVJjT2ayVORpqS7yiKhExi7NnQECntVm8haHaHtFNVDMT8X8b+htnixpXhPIQ9yV+ETrvr+hNeYfh+Ex+UpC//QPWnLcEOC8XOLgmQN8BNgRGSERWUziK0" + "bucket": "travis-php-db-import-tests-s3filesbucket-vm9zhtm5jd7s", + "key": "tw_accounts.csv", + "saveAs": "first-table", + "includeSubfolders": false, + "newFilesOnly": true + } +} +``` + +The first two parameters (`accessKeyId` and `#secretAccessKey`) are taken from the root `configuration`, the other +parameters are taken from the first rows' `configuration`. The `processors` node is never passed to the configuration file. +With the above configuration, the component will be executed only once, because there is one row. If there are no rows, the +component will still be executed once. If there were two rows, the component would be executed twice. + +If the component is executed more than once, the operations are executed in the following order: + +- input mapping for the first row +- run with the first row configuration (merged with root configuration) +- output mapping for the first row +- input mapping for the second row +- run with the second row configuration (merged with root configuration) +- output mapping for the second row + +All of these are executed in a single [job](/integrate/jobs/). However, even though multiple rows are executed in a single +job, the actual executions are still completely isolated. I.e., there is no way to share anything between the rows +(apart from the common `configuration`). It also means that the outputs of the first row are available in the Keboola project before +the second row starts, and the inputs for the second row are read only after the first row finishes processing. + +What is considered 'first' and 'second' -- i.e. the order of rows -- is defined by the order of items in the `rows` array. +See [below](#modifying-a-configuration) for an example of modifying the row order. + +Theoretically, configuration rows are supported for every component as long as the effective configuration matches what +the component expects. Configuration rows can be used to split the configuration into a common part (typically credentials) and an +iterable part which is repeated many times. Keep in mind that configurations heavily modified through the API might **not be supported +in the UI**. + +### State +The `state` node contains the content of the [state file](/extend/common-interface/config-file/#state-file). The +`state` is read from the state file and then supplied to the state file on the next run. In the above configuration, +the state is: + +```json +{ + "lastDownloadedFileTimestamp": "1511176959", + "processedFilesInLastTimestampSecond": [ + "tw_accounts.csv" + ] +} +``` + +`State` is considered an internal property of a component and you should avoid modifying it. The only reasonable modification of +`state` is to delete it -- in that case, the configuration will run as if it were run for the first time. To delete the `state`, set it to `{}`. +If configuration rows are used, then the `state` is stored separately for each row and the `state` node in configuration root is +not used. + +## Working with Configurations +Here, the most common operations done with configurations are described in examples. Feel free to go through the +[API reference](https://api.keboola.com/?service=storage#tag--Component-Configurations) for a full authoritative list of configuration features. + +### List Configurations +To obtain configuration details, use the [List Configs call](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs), +which will return all the configuration details. This means + +- the configuration itself (`configuration`) --- [section on configuration](#modifying-a-configuration) follows; +- configuration rows (`rows`) --- additional data of the configuration; and +- configuration state (`state`) --- [component state](/extend/common-interface/config-file/#state-file). + +Please note that the contents +of the `configuration`, `rows` and `state` sections depend purely on the component itself. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb). + +A sample result for the AWS S3 extractor looks like this: + +```json +[ + { + "id": "364479526", + "name": "test", + "description": "", + "created": "2018-03-08T14:54:19+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "version": 4, + "changeDescription": "Table first table edited", + "isDeleted": false, + "configuration": { + "parameters": { + "accessKeyId": "AKIAIBZYEEXQILP46FCA", + "#secretAccessKey": "KBC::ComponentProjectEncrypted==p5gvUw4RSGiVJjT2ayVORpqS7yiKhExi7NnQECntVm8haHaHtFNVDMT8X8b+htnixpXhPIQ9yV+ETrvr+hNeYfh+Ex+UpC//QPWnLcEOC8XOLgmQN8BNgRGSERWUziK0" + } + }, + "rowsSortOrder": [], + "rows": [ + { + "id": "364481153", + "name": "first table", + "description": "", + "configuration": {...}, + "isDisabled": false, + "version": 2, + "created": "2018-03-08T14:58:33+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited", + "state": {} + } + ], + "state": {}, + "currentVersion": { + "created": "2018-03-08T15:21:28+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited" + } + } +] +``` + +### Modifying Configuration +**Note: Configurations modified through the API might not be editable in the Keboola UI.** They can be run or used in an orchestration without any problems. + +Modifying a configuration means that a new version of that configuration is created. +For modifying a configuration, use the +[Update Configuration](https://api.keboola.com/?service=storage#put-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-) API call. +See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) in which the +configuration is modified to the following to set new credentials: + +```json +{ + "parameters": { + "accessKeyId": "a", + "#secretAccessKey": "b" + } +} +``` + +Notice that the configuration must be sent in the form field `configuration` as the endpoint does not accept pure JSON (yet). +Take great care to pass **only the contents** of the `configuration` node as in the above example. The configuration **must not be wrapped** in the +`configuration` node, otherwise the component will not +receive the configuration it expects. Also take care to properly escape the JSON using [URL encoding](https://en.wikipedia.org/wiki/Percent-encoding), +otherwise it may be misinterpreted. The raw HTTP request should look similar to this: + + curl --request PUT \ + --url https://connection.keboola.com/v2/storage/components/keboola.ex-aws-s3/configs/364479526 \ + --header "Content-Type: application/json" \ + --header 'X-StorageAPI-Token: {{token}}' \ + --data-binary "{ + \"configuration\": { + \"parameters\": { + \"accessKeyId\": \"a\", + \"#secretAccessKey\": \"b\" + } + } + }" + +Also note that the entire configuration must be always sent, there is no way to patch only part of it. +The same way the `configuration` is modified, other properties can be modified too. For example, you may want to +reset `state` by setting it to `{}`, or you can change the order of the configuration rows by setting the `rowsSortOrder` property. +The `rowsSortOrder` is an array of row ids -- see an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) (Set Row order of S3 extractor) +for the exact example request. + +### Modifying Configuration Row +Very similar to modifying a configuration, modifying a configuration **row** means that a new version of +the **entire configuration** is created. For modifying a configuration row, use the +[Update Row](https://api.keboola.com/?service=storage#put-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/rows/-rowId-) API call. + +See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) in which the +configuration row is modified to: + +```json +{ + "parameters": { + "bucket": "some-bucket", + "key": "sample.csv", + "includeSubfolders": false, + "newFilesOnly": true + } +} +``` + +The rules for updating a configuration row are the same as for [updating a configuration](#modifying-a-configuration). Also note that +a configuration row is never evaluated alone, it is always merged with the root `configuration`. If the same properties are defined +in the root `configuration` and row `configuration`, the values from the row are used. There is also an +[example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) of how to reset the row +state by setting `state` to `{}`. + +### Configuration Versions +When you [update a configuration](https://api.keboola.com/?service=storage#put-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-), +a new configuration version is actually created. In the above calls, only the last (active/published) configuration +is returned. To obtain a list of all recorded versions, use the +[List Versions API call](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/versions). +See this [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) +which would give you an output similar to the one below: + +```json +[ + { + "version": 4, + "created": "2018-03-08T15:21:28+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table edited", + "isDeleted": false, + "name": "test", + "description": "" + }, + { + "version": 3, + "created": "2018-03-08T14:58:33+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "Table first table added", + "isDeleted": false, + "name": "test", + "description": "" + }, + { + "version": 2, + "created": "2018-03-08T14:55:50+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "AWS Credentials edited", + "isDeleted": false, + "name": "test", + "description": "" + }, + { + "version": 1, + "created": "2018-03-08T14:54:19+0100", + "creatorToken": { + "id": 27865, + "description": "ondrej.popelka@keboola.com" + }, + "changeDescription": "", + "isDeleted": false, + "name": "test", + "description": "" + } +] +``` + +The field `version` represents the `version_id` in the following API example. + +### Rollback Configuration +After choosing a particular version, you can revert to that version by +[rolling back](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/versions/-versionId-/rollback), +i.e., making a new version identical to the chosen one. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D#2050856a-66b3-4120-9552-d1278a96621e) +of how to rollback the configuration `364479526` of the `keboola.ex-aws-s3` component to version `3`. + +It will create a new version of the configuration and return the ID of the version: +```json +{ + "version": "26" +} +``` + +### Creating Configuration Copy +After choosing a particular version, you can create a new independent +[configuration copy](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/versions/-versionId-/create) +of it. See an [example](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) +of how to create a new configuration called `test-copy` from version `3` of the `364479526` configuration +for the `keboola.ex-aws-s3` component. + +It will return the ID of the newly created configuration: +```json +{ + "id": "364494012" +} +``` + diff --git a/src/content/docs/integrate/storage/api/import-export/index.md b/src/content/docs/integrate/storage/api/import-export/index.md new file mode 100644 index 000000000..aa0dc68b9 --- /dev/null +++ b/src/content/docs/integrate/storage/api/import-export/index.md @@ -0,0 +1,340 @@ +--- +title: Manually Importing and Exporting Data +slug: 'integrate/storage/api/import-export' +--- + + +## Working with Data +Keboola Table Storage (Tables) and Keboola File Storage (File Uploads) are heavily connected together. +Keboola File Storage is technically a layer on top of the Amazon S3 service, and Keboola Table +Storage is a layer on top of a [database backend](/storage/#backends). + +To upload a table, take the following steps: + +- Request a [file upload](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/files/prepare) from +Keboola File Storage. You will be given a destination for the uploaded file on an S3 server. +- Upload the file there. When the upload is finished, the data file will be available in the *File Uploads* section. +- Initiate an [asynchronous table import](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async) +from the uploaded file (use it as the `dataFileId` parameter) into the destination table. +The import is asynchronous, so the request only creates a job and you need to poll for its results. +The imported files must conform to the [RFC4180 Specification](https://tools.ietf.org/html/rfc4180). + +![Schema of file upload process](/integrate/storage/api/async-import-handling.svg) + +Exporting a table from Storage is analogous to its importing. First, data is [asynchronously +exported](https://keboola.docs.apiary.io/#reference/tables/unload-data-asynchronously/asynchronous-export) from +Table Storage into File Uploads. Then you can request to [download +the file](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/files/-fileId-), which will give you +access to an S3 server for the actual file download. + +### Manually Uploading a File +To upload a file to Keboola File Storage, follow the instructions outlined in the +[API documentation](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/files/prepare). +First create a file resource; to create a new file called +[`new-file.csv`](/integrate/storage/new-table.csv) with `52` bytes, call: + +```bash +curl --request POST --header "Content-Type: application/json" --header "X-StorageApi-Token:storage-token" --data-binary "{ \"name\": \"new-file.csv\", \"sizeBytes\": 52, \"federationToken\": 1 }" https://connection.keboola.com/v2/storage/files/prepare +``` + +Which will return a response similar to this: + +```json +{ + "id": 192726698, + "created": "2016-06-22T10:44:35+0200", + "isPublic": false, + "isSliced": false, + "isEncrypted": false, + "name": "new_file2.csv", + "url": "https://s3.amazonaws.com/kbc-sapi-files/exp-15/1134/files/2016/06/22/192726697.new_file2?X-Amz-Content-Sha256=UNSIGNED-PAYLOAD&X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=AKIAJ2N244XSWYVVYVLQ%2F20160622%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date=20160622T084435Z&X-Amz-SignedHeaders=host&X-Amz-Expires=3600&X-Amz-Signature=86136cced74cdf919953cde9e2a0b837bd0b8f147aa6b7b30c2febde3b92d83d", + "region": "us-east-1", + "sizeBytes": 52, + "tags": [], + "maxAgeDays": 15, + "runId": null, + "runIds": [], + "creatorToken": { + "id": 53044, + "description": "ondrej.popelka@keboola.com" + }, + "uploadParams": { + "key": "exp-15/1134/files/2016/06/22/192726697.new_file2.csv", + "bucket": "kbc-sapi-files", + "acl": "private", + "credentials": { + "AccessKeyId": "ASI...H7Q", + "SecretAccessKey": "QbO...7qu", + "SessionToken": "Ago...bsF", + "Expiration": "2016-06-22T20:44:35+00:00" + } + } +} +``` + +The important parts are: `id` of the file, which will be needed later, the `uploadParams.credentials` node, +which gives you credentials to AWS S3 to upload your file, and +the `key` and `bucket` nodes, which define the target S3 destination as *s3://`bucket`/`key`*. +To upload the files to S3, you need an S3 client. There are a large number of clients available: +for example, use the +[S3 AWS command line client](https://docs.aws.amazon.com/cli/latest/userguide/cli-chap-install.html). +Before using it, [pass the credentials](https://docs.aws.amazon.com/cli/latest/topic/config-vars.html#credentials) +by executing, for instance, the following commands + +on *nix systems: +```bash +export AWS_ACCESS_KEY_ID=ASI...H7Q +export AWS_SECRET_ACCESS_KEY=QbO...7qu +export AWS_SESSION_TOKEN=Ago...wU= +``` + +or on Windows: +```bash +SET AWS_ACCESS_KEY_ID=ASI...H7Q +SET AWS_SECRET_ACCESS_KEY=QbO...7qu +SET AWS_SESSION_TOKEN=Ago...bsF +``` + +Then you can actually upload the `new-table.csv` file by executing the AWS S3 CLI [cp command](https://docs.aws.amazon.com/cli/latest/reference/s3/cp.html): +```bash +aws s3 cp new-table.csv s3://kbc-sapi-files/exp-15/1134/files/2016/06/22/192726697.new_file2.csv +``` + +After that, import the file into Table Storage, by calling either +[Create Table API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/buckets/-id-/tables-async) +(for a new table) or +[Load Data API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async) +(for an existing table). + +```bash +curl --request POST --header "Content-Type: application/json" --header "X-StorageApi-Token:storage-token" --data-binary "{ \"dataFileId\": 192726698, \"name\": \"new-table\" }" https://connection.keboola.com/v2/storage/buckets/in.c-main/tables-async +``` + +This will create an asynchronous job, importing data from the `192726698` file into the `new-table` destination table in the `in.c-main` bucket. +Then [poll for the job results](/integrate/jobs/#job-polling), or review its status in the UI. + +#### Python Example +The above process is implemented in the following example script in Python. This script uses the +[Requests](https://2.python-requests.org/en/master/) library for sending HTTP requests and +the [Boto 3](https://github.com/boto/boto3) library for working with Amazon S3. Both libraries can be +installed using pip: + +```bash +pip install boto3 +pip install requests +``` + +```python +import requests +import os +import json +import boto3 +from time import sleep + +storageToken = 'yourToken' +# Source filename (including path) +fileName = 'simple.csv' +# Target Storage Bucket (assumed to exist) +bucketName = 'in.c-main' +# Target Storage Table (assumed NOT to exist) +tableName = 'my-new-table' + +print('\nCreating upload file') + +# Create a new file in Storage +# See https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/files/prepare +response = requests.post( + 'https://connection.keboola.com/v2/storage/files/prepare', + data={ + 'name': fileName, + 'sizeBytes': os.stat(fileName).st_size, + 'federationToken': 1 + }, + headers={'X-StorageApi-Token': storageToken} +) +parsed = json.loads(response.content.decode('utf-8')) +# print(response.request.body) +# print(json.dumps(parsed, indent=4)) + +# Get AWS Credentials +accessKeyId = parsed['uploadParams']['credentials']['AccessKeyId'] +accessKeySecret = parsed['uploadParams']['credentials']['SecretAccessKey'] +sessionToken = parsed['uploadParams']['credentials']['SessionToken'] +region = parsed['region'] +fileId = parsed['id'] + +print('\nUploading to S3') + +# Upload file to S3 +# See https://boto3.amazonaws.com/v1/documentation/api/latest/guide/configuration.html +s3 = boto3.resource('s3', region_name=region, aws_access_key_id=accessKeyId, aws_secret_access_key=accessKeySecret, aws_session_token=sessionToken) +data = open(fileName, 'rb') +s3.Bucket(parsed['uploadParams']['bucket']).put_object(Key=parsed['uploadParams']['key'], Body=data) + +print('\nCreating table') + +# Load data from file into the Storage table +# See https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/buckets/-id-/tables-async +response = requests.post( + 'https://connection.keboola.com/v2/storage/buckets/%s/tables-async' % bucketName, + data={'name': tableName, 'dataFileId': fileId, 'delimiter': ',', 'enclosure': '"'}, + headers={'X-StorageApi-Token': storageToken}, +) +parsed = json.loads(response.content.decode('utf-8')) +# print(json.dumps(parsed, indent=4)) +if (parsed['status'] == 'error'): + print(parsed['error']) + exit(2) + +status = parsed['status'] +while (status == 'waiting') or (status == 'processing'): + print('\nWaiting for import to finish') + # See https://api.keboola.com/?service=storage#get-/v2/storage/jobs/-jobId- + response = requests.get(parsed['url'], headers={'X-StorageApi-Token': storageToken}) + jobParsed = json.loads(response.content.decode('utf-8')) + status = jobParsed['status'] + sleep(1) + +# print(json.dumps(jobParsed, indent=4)) +if (jobParsed['status'] == 'error'): + print(jobParsed['error']['message']) + exit(2) +``` + +#### Upload Files Using Storage API Importer +For production setup, we recommend using the approach [outlined above](#manually-uploading-a-file) +with direct upload to S3 as it is more reliable and universal. +In case you need to avoid using an S3 client, it is also possible to upload the +file by a simple HTTP request to [Storage API Importer Service](/integrate/storage/api/importer/). + +```bash +curl --request POST --header "X-StorageApi-Token:storage-token" --form "data=@new-file.csv" https://import.keboola.com/upload-file +``` + +The above will return a response similar to this: + +```json +{ + "id": 418137780, + "created": "2018-07-17T13:48:57+0200", + "isPublic": false, + "isSliced": false, + "isEncrypted": true, + "name": "404.md", + "url": "https:\/\/kbc-sapi-files.s3.amazonaws.com\/exp-15\/4088\/files\/2018\/07\/17\/418137779.new-file.csv...truncated", + "region": "us-east-1", + "sizeBytes": 1765, + "tags": [], + "maxAgeDays": 15, + "runId": null, + "runIds": [], + "creatorToken": { + "id": 144880, + "description": "file upload" + } +} +``` + +After that, import the file into Table Storage by calling either +[Create Table API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/buckets/-id-/tables-async) +(for a new table) or +[Load Data API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async) +(for an existing table). + +### Working with Sliced Files +Depending on the backend and table size, the data file may be sliced into chunks. +Requirements for uploading sliced files are described in the respective part of the +[API documentation](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/files/prepare). + +When you attempt to download a sliced file, you will instead obtain its manifest +listing the individual parts. Download the parts individually and join them +together. For a reference implementation of this process, see +our [TableExporter class](https://github.com/keboola/storage-api-php-client/blob/master/src/Keboola/StorageApi/TableExporter.php). + +**Important:** When exporting a table through the *Table* --- *Export* UI, the file will +be already merged and listed in the *File Uploads* section with the `storage-merged-export` tag. + +If you want to download a sliced file, [get credentials](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/files/-fileId-) +to download the file from AWS S3. Assuming that the file ID is 192611596, for example, call + +```bash +curl --header "X-StorageAPI-Token: storage-token" https://connection.keboola.com/v2/storage/files/192611596?federationToken=1 +``` + +which will return a response similar to this: + +```json +{ + "id": 192611596, + "created": "2016-06-21T15:25:35+0200", + "name": "in.c-redshift.blog-data.csv", + "url": "https://s3.amazonaws.com/kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csvmanifest?X-Amz-Content-Sha256=UNSIGNED-PAYLOAD&X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=AKIAJ2N244XSWYVVYVLQ%2F20160621%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date=20160621T135137Z&X-Amz-SignedHeaders=host&X-Amz-Expires=3600&X-Amz-Signature=ee69d94f0af06bcf924df0f710dcd92e6503a13c8a11a86be2606552bf9a8b26", + "region": "us-east-1", + "sizeBytes": 24541, + "tags": [ + "table-export" + ], + ... + "s3Path": { + "bucket": "kbc-sapi-files", + "key": "exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv" + }, + "credentials": { + "AccessKeyId": "ASI...UQQ", + "SecretAccessKey": "LHU...HAp", + "SessionToken": "Ago...uwU=", + "Expiration": "2016-06-22T01:51:37+00:00" + } +} +``` + +The field `url` contains the URL to the file manifest. Upon downloading it, you will get a JSON file with contents +similar to this: + +```json +{ + "entries": [ + {"url":"s3://kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv0000_part_00"}, + {"url":"s3://kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv0001_part_00"} + ] +} +``` + +Now you can download the actual data file slices. URLs are provided in the manifest file, and credentials to them +are returned as part of the previous file info call. To download the files from S3, you need an S3 client. There +are a wide number of clients available; for example, use the +[S3 AWS command line client](https://docs.aws.amazon.com/cli/latest/userguide/cli-chap-install.html). Before +using it, [pass the credentials](https://docs.aws.amazon.com/cli/latest/topic/config-vars.html#credentials) +by executing , for instance, the following commands + +on *nix systems: +```bash +export AWS_ACCESS_KEY_ID=ASI...UQQ +export AWS_SECRET_ACCESS_KEY=LHU...HAp +export AWS_SESSION_TOKEN=Ago...wU= +``` + +or on Windows: +```bash +SET AWS_ACCESS_KEY_ID=ASI...UQQ +SET AWS_SECRET_ACCESS_KEY=LHU...HAp +SET AWS_SESSION_TOKEN=Ago...wU= +``` + +Then you can actually download the files by executing the AWS S3 CLI [cp command](https://docs.aws.amazon.com/cli/latest/reference/s3/cp.html): +```bash +aws s3 cp s3://kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv0000_part_00 192611594.csv0000_part_00 +aws s3 cp s3://kbc-sapi-files/exp-2/578/table-exports/in/c-redshift/blog-data/192611594.csv0001_part_00 192611594.csv0001_part_00 +``` + +After that, merge the files together by executing the following commands + +on *nix systems: +```bash +cat 192611594.csv0000_part_00 192611594.csv0001_part_00 > merged.csv +``` + +or on Windows: +```bash +copy 192611594.csv0000_part_00 /B +192611594.csv0001_part_00 /B merged2.csv +``` diff --git a/src/content/docs/integrate/storage/api/importer/index.md b/src/content/docs/integrate/storage/api/importer/index.md new file mode 100644 index 000000000..063aaba5f --- /dev/null +++ b/src/content/docs/integrate/storage/api/importer/index.md @@ -0,0 +1,48 @@ +--- +title: Storage API Importer +slug: 'integrate/storage/api/importer' +--- + + +The [whole process of importing](/integrate/storage/api/) a table into Storage can be simplified with the +Storage API Importer Service. +The Storage API Importer allows you to make an HTTP POST request and import a file directly into an existing Storage table. + +The HTTP request must contain the `tableId` and `data` form fields. The specified table must already exist in [Storage](/storage/). +Therefore to upload the `my-table.csv` CSV file (and replace the contents) into the `my-table` table in the `in.c-main` bucket, +call: + +```bash +curl --request POST --header "X-StorageApi-Token:storage-token" --form "tableId=in.c-main.my-table" --form "data=@my-table.csv" "https://import.keboola.com/write-table" +``` + +Using the Storage API Importer is the easiest way to upload data into Storage (except for +using one of the [API clients](/integrate/storage/#clients)). However, the disadvantage is that the whole data file +has to be posted in a single HTTP request. **The maximum limit for a file size is 2GB and the transfer time is 45 minutes**. +This means that for substantially large files (usually more than hundreds of MB) +you may experience timeouts. If that happens, use the above outlined approach and upload the +files [directly to S3](/integrate/storage/api/import-export/#manually-uploading-a-file). + +## Parameters + +- `tableId` (required) Storage Table ID, example: in.c-main.users +- `data` (required) Uploaded CSV file. Raw file or compressed by [gzip](http://www.gzip.org/) +- `delimiter` (optional) Field delimiter used in a CSV file. The default value is ' , '. Use '\t' or type the tab char for tabulator. +- `enclosure` (optional) Field enclosure used in a CSV file. The default value is '"'. +- `escapedBy` (optional) CSV escape character; empty by default. +- `incremental` (optional) If incremental is set to 0 (its default), the target table is truncated before each import. + +Full list of avaialable parameters is available in the [API documentation](https://api.keboola.com/?service=import#import). + +## Examples +To load data incrementally (append new data to existing contents): + +```bash +curl --request POST --header "X-StorageApi-Token:storage-token" --form "incremental=1" --form "tableId=in.c-main.my-table" --form "data=@my-table.csv" "https://import.keboola.com/write-table" +``` + +To load data with a non-default delimiter (tabulator) and enclosure (empty): + +```bash +curl --request POST --header "X-StorageApi-Token:storage-token" --form "delimiter=\t" --form "enclosure=" --form "tableId=in.c-main.my-table" --form "data=@my-table.csv" "https://import.keboola.com/write-table" +``` diff --git a/src/content/docs/integrate/storage/api/index.md b/src/content/docs/integrate/storage/api/index.md new file mode 100644 index 000000000..18de8168d --- /dev/null +++ b/src/content/docs/integrate/storage/api/index.md @@ -0,0 +1,34 @@ +--- +title: Storage API +slug: 'integrate/storage/api' +--- + + +If you are new to Keboola, you should make yourself familiar with +the [Storage component](/storage/) before you start using it. +For a general introduction to working with Keboola APIs, see the [API Introduction](/overview/api/). +[Storage API](https://api.keboola.com/?service=storage) provides a number of functions. These are the most important ones: + +- [Component configurations](https://api.keboola.com/?service=storage#tag--Component-Configurations) +- [Storage tables](https://api.keboola.com/?service=storage#tag--Tables) +- [File uploads](https://api.keboola.com/?service=storage#tag--Files) +- [Storage buckets](https://api.keboola.com/?service=storage#tag--Buckets) + +Virtually, all API calls require a [Storage API token](/storage/tokens/) to +be passed as the `X-StorageApi-Token` header. +Please note that the Storage API calls require the request to be sent +as `form-data` (unlike the rest of Keboola API, which is sent as `application/json`). + +For exporting tables from and importing tables to Storage, we highly recommend that you use one of the +[available clients](/integrate/storage/) or the [Storage API Importer service](/integrate/storage/api/importer/). +All imports and exports are done using CSV files. See +the [RFC4180 Specification](https://tools.ietf.org/html/rfc4180) for the format +and encoding specification, and +[User documentation](/storage/tables/csv-files/) for help on how to create such files. + +Continue reading the following sections for guidance on how to get started: + +- [Storage importer service for the easiest upload of data via API](/integrate/storage/api/importer/) +- [Getting started with component configurations](/integrate/storage/api/configurations/) +- [Importing and exporting data](/integrate/storage/api/import-export/) +- [TDE exporter for exporting data to Tableau Data Extracts](/integrate/storage/api/tde-exporter/) diff --git a/src/content/docs/integrate/storage/api/tde-exporter/index.md b/src/content/docs/integrate/storage/api/tde-exporter/index.md new file mode 100644 index 000000000..33aaf8204 --- /dev/null +++ b/src/content/docs/integrate/storage/api/tde-exporter/index.md @@ -0,0 +1,93 @@ +--- +title: TDE Exporter +slug: 'integrate/storage/api/tde-exporter' +--- + + +[TDE Exporter](https://github.com/keboola/tde-exporter) exports tables from Keboola Storage into the +[TDE file format (Tableau Data Extract)](https://www.tableau.com/about/blog/2014/7/understanding-tableau-data-extracts-part1). +This component is normally a part of the [Tableau Writer](/tutorial/write/), +but it can also be used as a standalone component. + +Users can [run a TDE exporter job](/integrate/jobs/) as any other Keboola component or register it +as an orchestration task. After the exporter finishes, the resulting TDE files will be available in the +*Storage* --- *File uploads* section where you can download them via UI or [API](/integrate/storage/api/import-export/). + +## Running the Component +The TDE Exporter is a Keboola [component](/extend/component/) supporting both +[stored](/integrate/storage/api/configurations/) and +custom configurations supplied directly in the `run` request. + +### Stored Configuration +To run the TDE exporter with a stored configuration, first +[create the configuration](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs). +See [below](#custom-configuration) for the required configuration contents. +This call will give you the ID of the newly created configuration (for instance, `new-configuration-id`). +Then [create a job](/integrate/jobs/) with the specified configuration: + +```json +{ + "config": "new-configuration-id" +} +``` + +### Custom Configuration +You can specify the entire configuration in the API call. The JSON configuration conforms +to the [general configuration format](/extend/common-interface/config-file/). The specific part +is only the `parameters` section. A sample request to the `in.c-main.old-table` export table would look like this: + +```json +{ + "configData": { + "storage": { + "input": { + "tables": [{ + "source": "in.c-main.old-table" + }] + } + }, + "parameters": { + "tags": ["sometag"], + "typedefs": { + "in.c-main.old-table": { + "id": { + "type": "number" + }, + "col1": { + "type": "string" + } + } + } + } + } +} +``` + +The `parameters` section contains: + +- `tags`: array of tags that will be assigned to the resulting file in Storage File Uploads. +- `typedefs`: definitions of data types mapping source tables columns to destination TDE columns. + +The type definitions are entered as an object whose name must match the name of the table in the +`storage.input.tables.source` node (`in.c-main.old-table` in the above example). Object properties +are names of the table columns; each must have the `type` property which is one of the +[supported column types](https://help.tableau.com/current/pro/desktop/en-us/datafields_typesandroles_datatypes.htm): +`boolean`, `number`, `decimal`, `date`, `datetime` and `string`. + +## Date and DateTime +Data for these data types can be specified in the format used +in the [strptime function](https://pubs.opengroup.org/onlinepubs/009695399/functions/strptime.html). The format is specified as part of the column's type definition. For example: + +```json +{ + "col1": { + "type": "date", + "format":"%m-%d-%Y" + } +} +``` + +If no format is specified, the following default formats are used: + +- For `date`: `%Y-%m-%d` +- For `datetime`: `%Y-%m-%d %H:%M:%S or %Y-%m-%d %H:%M:%S.%f` diff --git a/src/content/docs/integrate/storage/docker-cli-client/index.md b/src/content/docs/integrate/storage/docker-cli-client/index.md new file mode 100644 index 000000000..4b4d7a2fb --- /dev/null +++ b/src/content/docs/integrate/storage/docker-cli-client/index.md @@ -0,0 +1,111 @@ +--- +title: Storage Docker CLI Client +slug: 'integrate/storage/docker-cli-client' +redirect_from: + - /integrate/storage/php-cli-client/ +--- + + +The Storage API Docker command line interface (CLI) client is a portable command line client which provides +a simple implementation of [Storage API](https://api.keboola.com/?service=storage). +It runs on any platform which has Docker installed. + +Currently, the client implements + +- functions for exporting and importing tables; +- functions for creating and deleting buckets; and additionally, +- the [project backup feature](/management/project-export/). + +The client source is available in our [Github repository](https://github.com/keboola/storage-api-cli). +The client docker image is available in the [Quay repository](https://quay.io/repository/keboola/storage-api-cli?tab=tags). + +## Running in Docker +To print available commands: + +```bash +docker run quay.io/keboola/storage-api-cli:latest +``` + +The `latest` image tag always refers to the latest tagged version. + +## Running Phar + +PHAR (PHP Archive) is now deprecated, but there are still some older versions available. See the [repository documentation](https://github.com/keboola/storage-api-cli#running-phar-deprecated). + +### Example --- Creating a Table +To create a new table in Storage, use the `create-table` command. Provide the name of an +existing bucket, the name of the new table and a CSV file with the table's contents. + +To create the`new-table` table in the `in.c-main` bucket, use + +```bash +docker run --volume=$("pwd"):/data quay.io/keboola/storage-api-cli:latest create-table in.c-main new-table /data/new-table.csv --token=storage_token +``` + +or on Windows: + + docker run --volume=C:\Users\name\some-dir:/data quay.io/keboola/storage-api-cli:latest create-table in.c-main new-table /data/new-table.csv --token=storage_token + +or when using other then [default US region](/overview/api/#regions-and-endpoints), you need to provide the Storage API address: + +```bash +docker run --volume=$("pwd"):/data quay.io/keboola/storage-api-cli:latest create-table in.c-main new-table /data/new-table.csv --token=storage_token --url="https://connection.eu-central-1.keboola.com/" +``` + +Any of the above commands will import the contents of `new-table.csv` in the current directory into the newly +created table. You should see an output similar to this one: + + Authorized as: ondrej.popelka@keboola.com (Odinuv Sandbox) + Bucket found ok + Table create start + Table create end + Table id: in.c-main.new-table + +*Please note that the Docker container can only access folders within the container, so you need to mount a local folder. +In the example above, the local folder `$("pwd")` (replaced by the absolute path at runtime) is mounted as `/data` into the container. +The table is then accessible in this folder. The same approach applies to all other commands working with local files.* + +### Example --- Importing Data +If you only want to import new data into the table, use the `write-table` command and provide +the ID (*bucketName.tableName*) of an existing table. + +To import data into the `new-table` table in the `in.c-main` bucket, use + +```bash +docker run --volume=$("pwd"):/data quay.io/keboola/storage-api-cli:latest write-table in.c-main.new-table /data/new-data.csv --token=storage_token --incremental +``` + +The above command will import the contents of the `new-data.csv` file into the existing table. If the +`--incremental` parameter is supplied, the table contents will be appended. If the parameter is not +supplied, the table contents will be overwritten. You should see an output similar to this one: + + Authorized as: ondrej.popelka@keboola.com (Tutorial) + Table found ok + Import start + Import done in 17 secs. + + Results: + transaction: + warnings: + importedColumns: + - id + - secondCol + totalRowsCount: 8 + totalDataSizeBytes: 4096 + +### Example --- Exporting Data +If you want to export a table from Storage, use the `export-table` command. Provide +the ID (*bucketName.tableName*) of an existing table. + +To export data from the `old-table` table in the `in.c-main` bucket, use + +```bash +docker run --volume=$("pwd"):/data quay.io/keboola/storage-api-cli:latest export-table in.c-main.old-table /data/old-data.csv --token=storage_token +``` + +The above command will export the table from Storage and save it as `old-data.csv` in +the current directory. You should see an output similar to this one: + + Authorized as: ondrej.popelka@keboola.com (Tutorial) + Table found ok + Export done in 17 secs. diff --git a/src/content/docs/integrate/storage/index.md b/src/content/docs/integrate/storage/index.md new file mode 100644 index 000000000..f31b28f1b --- /dev/null +++ b/src/content/docs/integrate/storage/index.md @@ -0,0 +1,56 @@ +--- +title: Storage +slug: 'integrate/storage' +--- + + +As the central Keboola component, Storage + +- Keeps all data in [**buckets** and **tables**](/storage/); +- Controls access to the data using **tokens**; +- Logs all data manipulations as **events**; +- Maintains the index of all other Keboola **components** and stores their **configurations**. + +All this (and a few other things) is available through [Storage API (SAPI)](https://api.keboola.com/?service=storage). +To authorize access to a specific project, most calls to Storage API require +a [Storage API Token](/storage/tokens/) along with your request. +It is required regardless of whether you use the bare API or any of the clients. + +## Storage API Clients +Although you can work directly with the API, we recommend using one of our Storage API clients, as they simplify some tasks. +There are four Storage clients with different feature sets available: + +1. [PHP client library](https://github.com/keboola/storage-api-php-client) --- a PHP library supporting most of the Storage API features; +use it programmatically in PHP. +2. [R client library](/integrate/storage/r-client/) --- an R library supporting most data manipulation features of the Storage API; +use it programmatically in R. +3. [Python client library](/integrate/storage/python-client/) --- a Python library supporting most data manipulation features and +workspace manipulation features of the Storage API; use it programmatically in Python. +4. [Docker CLI client](https://github.com/keboola/storage-api-cli) --- a CLI (command line interface) application supporting +basic data manipulation features of the Storage API; use it from the command line provided that you have Docker available. + +Additional tools: + +- [Storage API Console](https://storage-api-console.keboola.com/) --- a UI to work with Keboola Storage; +this is accessible to anyone with a Storage Token (not necessarily a Keboola project administrator) +- [Table Importer Service](/integrate/storage/api/importer/) --- a service designed for simplified table loads + +The client choice is purely up to you, but it is best to use the most straightforward solution. + +## Table Imports and Exports +Tables are imported to and exported from Storage via asynchronous (background) jobs. + +- When importing a table, the actual data is first transported to an Amazon S3 storage, +and then bulk is loaded into the internal database in Storage. Similarly, +- When exporting a table, the data is first offloaded to an Amazon S3 storage and downloaded from there. + +While this process is much more complicated than a simple file upload or download, +it offers better **manageability** and **traceability** features. +Use one of the above mentioned clients to import and export your data. The client will handle the entire process +without you worrying about the technical details. + +If all you need is to import data into Storage (for example, for project prototyping), you may +also use the [Storage Importer Service](/integrate/storage/api/importer/). + +Still interested in handling the file uploads/downloads manually? +[Read on](/integrate/storage/api/import-export/). diff --git a/src/content/docs/integrate/storage/php-client/index.md b/src/content/docs/integrate/storage/php-client/index.md new file mode 100644 index 000000000..f5c995d7b --- /dev/null +++ b/src/content/docs/integrate/storage/php-client/index.md @@ -0,0 +1,146 @@ +--- +title: Storage PHP Client Library +slug: 'integrate/storage/php-client' +--- + + +The Storage API PHP client library is a portable command line client providing +the most complete [Storage API](https://api.keboola.com/?service=storage) implementation. +It runs on any platform which has PHP installed. +Currently this client implements almost all Storage API functions including, of course, exporting and importing tables. + +The client source is available in our [Github repository](https://github.com/keboola/storage-api-php-client). + +## Installation + +The Library is available as a [Composer package](https://getcomposer.org/). +Unless you already have it, [install Composer](https://getcomposer.org/download/) on your system. +On *nix system, do so by running + +```bash +curl -s http://getcomposer.org/installer | php +mv ./composer.phar ~/bin/composer # or /usr/local/bin/composer +``` + +On Windows, use the [installer](https://getcomposer.org/Composer-Setup.exe). + +To install the library, run + +```bash +composer require keboola/storage-api-client +``` + +in the root of your project. You should get an output similar to this one: + + Using version ^4.11 for keboola/storage-api-client + ./composer.json has been created + Loading composer repositories with package information + Updating dependencies (including require-dev) + - Installing aws/aws-sdk-php (3.18.18) + Downloading: 100% + ... + - Installing keboola/storage-api-client (4.11.0) + Downloading: 100% + Writing lock file + Generating autoload files + +Then add the generated autoloader in your bootstrap script: + +```php +require 'vendor/autoload.php'; +``` + +You can read more in the [Composer documentation](https://getcomposer.org/doc/01-basic-usage.md). Packages +installable by Composer can be browsed at [Packagist package repository](https://packagist.org/). + +## Usage +The Storage API client is implemented as a single class. To create an instance of the class, provide a Storage API token to the +constructor. + +```php + 'your-token', +]); +``` + +### Example --- Create a Table +To create a new table in Storage, it is recommended to use an additional +[php-csv](https://github.com/keboola/php-csv) library to work +with CSV files. The library will get installed +automatically with the Storage API client, so you can use it out of the box. +To create a new table and import CSV data in it, use the following PHP script: + +```php + 'your-token', +]); +$csvFile = new CsvFile('./new-table.csv'); +$client->createTableAsync('in.c-main', 'new-table', $csvFile); +``` + +### Example --- Import Data +To import CSV data into an existing table and overwrite its contents, use the following PHP script: + +```php + 'your-token', +]); +$csvFile = new CsvFile('./new-table.csv'); +$client->writeTableAsync('in.c-main.new-table', $csvFile); +``` + +### Example --- Import Data Incrementally +To import CSV data into an existing table and append the new data to the existing table contents, use the following PHP script: + +```php + 'your-token', +]); +$csvFile = new CsvFile('./new-table.csv'); +$client->writeTableAsync('in.c-main.new-table', $csvFile, ['incremental' => true]); +``` + +All available upload options are listed in the [API documentation](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/tables/-id-/import-async). + +### Example --- Export Data +To export data from a Storage table to a CSV file, use the +`TableExporter` class. It is part of the client library. You can use the following script: + +```php + 'your-token' +]); + +$exporter = new TableExporter($client); +$exporter->exportTable('in.c-main.my-table', './old-table.csv'); +``` diff --git a/src/content/docs/integrate/storage/python-client/index.md b/src/content/docs/integrate/storage/python-client/index.md new file mode 100644 index 000000000..613a35a31 --- /dev/null +++ b/src/content/docs/integrate/storage/python-client/index.md @@ -0,0 +1,116 @@ +--- +title: Python Client Library +slug: 'integrate/storage/python-client' +--- + + +The Python client library is a [Storage API client](https://api.keboola.com/?service=storage) which you can use in your Python code. +The current implementation supports all basic data manipulations: + +- Importing data +- Exporting data +- Creating and deleting buckets and tables +- Creating and deleting workspaces + +The client source code is available in our [Github repository](https://github.com/keboola/sapi-python-client/). + +## Installation +This library is available on [Github](https://github.com/keboola/sapi-python-client), so we +recommend that you use the `pip` package to install it: + + pip3 install git+https://github.com/keboola/sapi-python-client.git + +## Usage +The client contains a `Client` class, which encapsulates all API endpoints and holds a storage token and URL. Each API endpoint is +represented by its own class (`Files`, `Buckets`, `Jobs`, etc.), which can be used standalone if you only work with one endpoint. +This means that the two following examples are equivalent: + +```python +from kbcstorage.client import Client + +client = Client('https://connection.keboola.com', 'your-token') +client.tables.detail('in.c-demo.some-table') +``` + +```python +from kbcstorage.tables import Tables + +tables = Tables('https://connection.keboola.com', 'your-token') +tables.detail('in.c-demo.some-table') +``` + +### Example --- Create Table and Import Data +To create a new table in Storage, use the `create` function of the `Tables` class. Provide the name of an existing bucket, +the name of the new table and a CSV file with the table's contents. + +To create the `new-table` table in the `in.c-main` bucket, use: + +```python +from kbcstorage.client import Client + +client = Client('https://connection.keboola.com', 'your-token') +client.tables.create(name='new-table', + bucket_id='in.c-main', + file_path='coords.csv', + primary_key=['id']) +``` + +The above command will import the contents of the `coords.csv` file into the newly created table. It will +also mark the `id` column as the primary key. +### Example --- Load to existing table, incrementally + +To load data incrementally into an existing table, we can use the [load](https://github.com/keboola/sapi-python-client/blob/5a93926c2191ccd6b7402c9e24d9912884d87d4c/kbcstorage/tables.py#L207) method, where `table_id` is the ID of the table that you want to load into, and `path` is the path to your csv file containing the data: + +```python + +from kbcstorage.client import Client + +client = Client('https://connection.keboola.com', 'your-token') + +client.tables.load(table_id=table_id, file_path=path, is_incremental=True) + +``` +### Example --- Export Data +To export data from the `old-table` table in the `in.c-main` bucket, use: + +```python +from kbcstorage.client import Client +import csv + +client = Client('https://connection.keboola.com', 'your-token') +client.tables.export_to_file(table_id='in.c-main.new-table', path_name='.') +with open('./new-table', mode='rt', encoding='utf-8') as in_file: + lazy_lines = (line.replace('\0', '') for line in in_file) + reader = csv.reader(lazy_lines, lineterminator='\n') + for row in reader: + print(row) +``` + +The above command will export the table from Storage into the file `new-table` and read it using +[CSV Reader](https://docs.python.org/3.6/library/csv.html#reader-objects). + +### Other Examples + +```python +# create a client +client = Client('https://connection.keboola.com', 'your-token') + +# create a bucket +client.buckets.create(name='demo', stage='in') + +# list buckets +client.buckets.list() + +# list all tables +client.tables.list() + +# list all tables in a bucket +client.buckets.list_tables(bucket_id='in.c-demo') + +# delete a table +client.tables.delete(table_id='in.c-demo.some-table') + +# delete a bucket +client.buckets.delete(bucket_id='in.c-main', force=True) + +``` diff --git a/src/content/docs/integrate/storage/r-client/index.md b/src/content/docs/integrate/storage/r-client/index.md new file mode 100644 index 000000000..e1c5b3c84 --- /dev/null +++ b/src/content/docs/integrate/storage/r-client/index.md @@ -0,0 +1,121 @@ +--- +title: R Client Library +slug: 'integrate/storage/r-client' +--- + + +The R client library is a [Storage API client](https://api.keboola.com/?service=storage) which you can use in your R code. +The current implementation supports all basic data manipulations: + +- Importing data +- Exporting data +- Creating and deleting buckets and tables + +The client source code is available in our [Github repository](https://github.com/keboola/sapi-r-client). + +## Installation +This library is available on [Github](https://github.com/keboola/sapi-r-client), so we +recommend that you use the `devtools` package to install it. + +```r +# first install the devtools package if it isn't already installed +install.packages("devtools") + +# install dependencies (another github package for aws requests) +devtools::install_github("cloudyr/aws.s3") + +# install the SAPI R client package +devtools::install_github("keboola/sapi-r-client") + +# load the library (dependencies will be loaded automatically) +library(keboola.sapi.r.client) +``` + +## Usage +To list available commands, run +```r +?keboola.sapi.r.client::SapiClient +``` + +**Important**: If you are running the code in R Studio, it might require a restart so that its help index is updated +and the above command works. + +The client is implemented as an [RC class](http://adv-r.had.co.nz/R5.html). To work with it, create an instance of the client. +The only required argument to create it is a valid Storage API token. + +```r +client <- SapiClient$new( + token = 'your-token' +) +``` + +### Example --- Create a Table and Import Data +To create a new table in Storage, use the `saveTable` function. Provide the name of an existing bucket, +the name of the new table and a CSV file with the table's contents. + +To create the `new-table` table in the `in.c-main` bucket, use + +```r +myDataFrame <- data.frame(id = c(1,2,3,4), secondCol = c('a', 'b', 'c', 'd')) +client <- SapiClient$new( + token = 'your-token' +) + +table <- client$saveTable( + df = myDataFrame, + bucket = "in.c-main", + tableName = "new-table", + options = list(primaryKey = 'id') +) +``` + +The above command will import the contents of the `myDataFrame` variable into the newly created table. It will +also mark the `id` column as the primary key. + +### Example --- Export Data +If you want to export a table from Storage and import it into R, use the `importTable` function. Provide +the ID (*bucketName.tableName*) of an existing table. + +To export data from the `old-table` table in the `in.c-main` bucket, use + +```r +client <- SapiClient$new( + token = 'your-token' +) + +data <- client$importTable('in.c-main.old-table') +``` + +The above command will export the table from Storage and save it in the `data` variable. The output is +a [data.table](https://cran.r-project.org/web/packages/data.table/index.html) object compatible with a `data.frame`. + +### Other Examples + +```r +# create a client +client <- SapiClient$new( + token = 'your-token' +) + +# verify the token +tokenDetails <- client$verifyToken() + +# create a bucket +bucket <- client$createBucket("new_bucket", "in", "A brand new Bucket!") + +# list buckets +buckets <- client$listBuckets() + +# list all tables +tables <- client$listTables() + +# list all tables in a bucket +tables <- client$listTables(bucket = bucket$id) + +# delete a table +client$deleteTable(table$id) + +# delete a bucket +client$deleteBucket(bucket$id) + +``` diff --git a/src/content/docs/integrate/variables/index.md b/src/content/docs/integrate/variables/index.md new file mode 100644 index 000000000..ca01fc258 --- /dev/null +++ b/src/content/docs/integrate/variables/index.md @@ -0,0 +1,725 @@ +--- +title: Variables +slug: 'integrate/variables' +--- + + +*Note: This is a preview feature and as such may change considerably in the future.* + +**Variables** are placeholders used in [configurations](/integrate/storage/api/configurations/). Their value is +resolved at [job runtime](/integrate/jobs/). + +**Important:** Make sure you're familiar with the [Configuration API](/integrate/storage/api/configurations/) and +the [Job API](/integrate/jobs/) before reading on. + +See [Tutorial](/integrate/variables/tutorial) for step-by-step example. + +## Introduction +When using variables, the configuration is treated as a [Moustache template](https://mustache.github.io/mustache.5.html). +You can enter variables anywhere in the JSON of the configuration body. The configuration body is the contents of +the `configuration` node when you [retrieve a configuration](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-). +This means that you can't use variables in a name or in a configuration description. + +Variables are entered using the [Moustache syntax](https://mustache.github.io/mustache.5.html), +i.e., `{{ variableName}}`. To work with variables, three things are needed: + +- Main configuration -- the configuration in which variables are replaced (used); this can be a configuration of any component (e.g., a configuration of a transformation, extractor, writer, etc.). +- Variable configuration -- a configuration in which variables are defined; this is a configuration of a special `keboola.variables` component. +- Variable values -- actual values that will be placed in the main configuration. + +To enable replacement of variables, the *main configuration* has to reference the *variable configuration*. +If there is no *variable configuration* referenced, no replacement is made (the *main configuration* is completely +static). Variables can be used in any place of any configuration except legacy transformations (the component with +the ID `transformation`; it can still be used in a specific transformation -- e.g., `keboola.python-transformation-v2` +or `keboola.snowflake-transformation`, etc.), and an orchestrator (see [below](#orchestrator-integration)). + +## Variable Configuration +A *variable configuration* is a standard configuration tied to a special dedicated `keboola.variables` component. +The variable configuration defines names of variables to be replaced in the main configuration. You can create +the configuration using the +[Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs). +This is an example of the contents of such a configuration: + +```json +{ + "variables": [ + { + "name": "firstVariable", + "type": "string" + }, + { + "name": "secondVariable", + "type": "string" + } + ] +} +``` + +Note that `type` is always `string`. + +## Main Configuration +When you create a variable configuration, you'll obtain an ID of the configuration - e.g., `807940806`. +In the *main configuration*, you have to reference the *variable configuration* ID using the `variables_id` node. +Then you can use the variables in the configuration body: + +```json + +{ + "storage": { + "input": { + "tables": [ + { + "source": "in.c-application-testing.{{firstVariable}}", + "destination": "{{firstVariable}}.csv" + } + ] + }, + "output": { + "tables": [ + { + "source": "new-table.csv", + "destination": "out.c-transformation-test.cars" + } + ] + } + }, + "parameters": { + "script": [ + "print('{{firstVariable}}')" + ] + }, + "variables_id": "807940806" +} + +``` + +## Variable Values +You can either store the variable values as [configuration rows](/integrate/storage/api/configurations/#configuration-rows) of the +*variable configuration* and provide the row ID of the stored values at run time, or you can provide the variable values directly at run +time. There are three options how you can provide values to the variables: + +- Reference values using `variables_values_id` property in the *main configuration* (default values). +- Reference values using `variablesValuesId` property in job parameters. +- Provide values using `variableValuesData` property in job parameters. + +The structure of variable values, regardless of whether it is stored in configuration or provided at runtime, is as follows: + +```json +{ + "values": [ + { + "name": "firstVariable", + "value": "batman" + } + ] +} +``` + +## Variable Delimiter +The default variable delimiter is `{{` and `}}`. If the delimiter interferes with your code, it can +be changed as per the [Moustache docs](https://mustache.github.io/mustache.5.html). For example the +following piece of code + +```json +{ + "code": "SELECT \"COUNTRY\" || '{{ alias}}' || '{{=<< >>=}} {{ as-is}} <<={{}}=>>' + AS \"COUNTRY\", \"CARS\" || '{{ size}}' AS \"CARS\" FROM \"my-table\"" +} +``` + +will be interpreted as (assuming the variables `alias=batman` and `size=big` are defined): + +```json +{ + "code": "SELECT \"COUNTRY\" || 'batman' || '{{ as-is}}' + AS \"COUNTRY\", \"CARS\" || 'big' AS \"CARS\" FROM \"my-table\"" +} +``` + +## Example Using Python Transformations +In this example, we will configure a Python transformation using variables. + +### Step 1 -- Create Variable Configuration +Use the [Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) +for the `keboola.variables` component with the following content: + +```json +{ + "variables": [ + { + "name": "alias", + "type": "string" + }, + { + "name": "size", + "type": "string" + } + ] +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#16a5d721-b6a4-4daa-9196-8e90250ed16b). + +### Step 2 -- Create Default Values for Variables +Note that this step is optional -- you can use variables without default values. +In the previous step, you obtained an ID of the variable configuration. Use the +[Create Configuration Row API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs/-configurationId-/rows). +Use the ID of the variable configuration and `keboola.variables` as a component. Use the following body: + +```json +{ + "values": [ + { + "name": "alias", + "value": "batman" + }, + { + "name": "size", + "value": "42" + } + ] +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#72de2851-1853-4fe3-bec3-856fcc9e2270) + +### Step 3 -- Create Main Configuration +Now it is time to create the actual configuration which will contain a Python transformation. +Use the following configuration body. The `storage` section describes the standard [input](/extend/common-interface/config-file/#input-mapping--basic) +and [output](/extend/common-interface/config-file/#output-mapping--basic) mapping. + +```json + +{ + "storage": { + "input": { + "tables": [ + { + "source": "in.c-variable-testing.{{alias}}", + "destination": "{{alias}}.csv" + } + ] + }, + "output": { + "tables": [ + { + "source": "new-table.csv", + "destination": "out.c-variable-testing.cars" + } + ] + } + }, + "parameters": { + "blocks": [ + { + "name": "First Block", + "codes": [ + { + "name": "First Code", + "script": [ + "import csv\ncsvlt = '\\n'\ncsvdel = ','\ncsvquo = '\"'\nwith open('in/tables/{{alias}}.csv', mode='rt', encoding='utf-8') as in_file, open('out/tables/new-table.csv', mode='wt', encoding='utf-8') as out_file:\n writer = csv.DictWriter(out_file, fieldnames=['COUNTRY', 'CARS'], lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo)\n writer.writeheader()\n\n lazy_lines = (line.replace('\\0', '') for line in in_file)\n reader = csv.DictReader(lazy_lines, lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo)\n for row in reader:\n writer.writerow({'COUNTRY': row['COUNTRY'] + '{{ alias }}', 'CARS': row['CARS'] + '{{ size }}'})\nfrom pathlib import Path\nimport sys\ncontents = Path('/data/config.json').read_text()\nprint(contents, file=sys.stdout)" + ] + } + ] + } + ] + } + "variables_id": "807968875", + "variables_values_id": "807952812" +} + +``` + +The `variables_id` property contains the ID of the [variable configuration](/integrate/variables/#step-1--create-variables-configuration) - e.g., `807968875`. The +`variables_values_id` property is optional and contains the ID of the [row with default values](/integrate/variables/#step-2--create-default-values-for-variable) - e.g., `807952812`. +The `parameters` section contains a script with the following Python code: + +```python + +import csv +csvlt = '\n' +csvdel = ',' +csvquo = '"' +with open('in/tables/{{alias}}.csv', mode='rt', encoding='utf-8') as in_file, open('out/tables/new-table.csv', mode='wt', encoding='utf-8') as out_file: + writer = csv.DictWriter(out_file, fieldnames=['COUNTRY', 'CARS'], lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo) + writer.writeheader() + lazy_lines = (line.replace('\0', '') for line in in_file) + reader = csv.DictReader(lazy_lines, lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo) + for row in reader: + writer.writerow({'COUNTRY': row['COUNTRY'] + '{{ alias }}', 'CARS': row['CARS'] + '{{ size }}'}) + +from pathlib import Path +import sys +contents = Path('/data/config.json').read_text() +print(contents, file=sys.stdout) + +``` + +The script reads a file given by the alias, modifies the two columns **COUNTRY** and **CARS**, and +prints the contents of the configuration file to output. + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#732e4b66-4f2d-46ab-80ba-7a7d07ddb94b). + +### Step 4 -- Run Job +There are three options for providing variable values when running a job: + +- Relying on default variables +- Providing ID of values using the `variablesValuesId` property in job parameters +- Providing values using the `variableValuesData` property in job parameters + +Following the rules for running a job, you always **have to** provide values for the defined variables. +Note that it is important which variables are *defined* in the variable configuration, not which +variables you actually use in the main configuration. For example, the main configuration references a variable +configuration with *firstVar* and *secondVar* variables, but you're using `{{ firstVar}}` and +`{{ thirdVar}}` in the configuration code. Then you have to provide values at least for *firstVar* +and *secondVar* variables. If you provide values for all *firstVar*, *secondVar*, and *thirdVar*, all of them will +be replaced. If you omit *thirdVar*, it will be replaced by an empty string. If you omit one of *firstVar*, +*secondVar*, an error will be raised. + +The second rule is that the three options of passing values are mutually exclusive. If you provide values using +`variablesValuesId` or `variableValuesData`, it overrides the default values (if provided). You can't use +`variablesValuesId` and `variableValuesData` together in a single call. If you do that, an error will be raised. +If no default values are set and none of the `variablesValuesId` or `variableValuesData` is provided, an error +will be raised. + +#### Option 1 -- Rely on default variables +If you created the default values, you can now directly run the job. Use the [Create Job API call](https://api.keboola.com/?service=job-queue#job-queue/tag/jobs/POST/jobs) +with the following body: + +```json +{ + "component": "keboola.python-transformation-v2", + "config": "807943784", + "mode": "run" +} +``` + +The `config` property contains the ID of the [main configuration](/integrate/variables/#step-3--create-main-configuration). +Before executing the API call, you have to create the source table. Unless you modified the mapping in the +[example](/integrate/variables/#step-3--create-main-configuration), you have to create a bucket named +**variable-testing** in the **in** stage. Then create a table called **batman** with columns **COUNTRY** +and **CARS**. You can use this [sample CSV file](/integrate/variables/countries.csv). + +After you create the input table, you can run the job. +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#31486ac2-ea52-4f19-a039-2ee1b1ae5863). +It will create a new table in Storage -- **out.c-variable-testing.cars**. The tables should contain the default +values, e.g.: + +|COUNTRY|CARS| +|---|---| +|Belgiumbatman|629378142| +|Finlandbatman|335823242| +|Italybatman|4139387742| +|Romaniabatman|654126042| + +The events of the job will contain the contents of the [configuration file](/extend/common-interface/config-file/) +where you can verify that the variables were replaced. + +
+ Click to expand the configuration. +```json +{ + "storage": { + "input": { + "tables": [ + { + "source": "in.c-variable-testing.batman", + "destination": "batman.csv", + "columns": [], + "where_values": [], + "where_operator": "eq" + } + ], + "files": [] + }, + "output": { + "tables": [ + { + "source": "new-table.csv", + "destination": "out.c-variable-testing.cars", + "incremental": false, + "primary_key": [], + "columns": [], + "delete_where_values": [], + "delete_where_operator": "eq", + "delimiter": ",", + "enclosure": "\"", + "metadata": [], + "column_metadata": [] + } + ], + "files": [] + } + }, + "parameters": { + "blocks": [ + { + "name": "First Block", + "codes": [ + { + "name": "First Code", + "script": [ + "import csv\ncsvlt = '\\n'\ncsvdel = ','\ncsvquo = '\"'\nwith open('in/tables/{{alias}}.csv', mode='rt', encoding='utf-8') as in_file, open('out/tables/new-table.csv', mode='wt', encoding='utf-8') as out_file:\n writer = csv.DictWriter(out_file, fieldnames=['COUNTRY', 'CARS'], lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo)\n writer.writeheader()\n\n lazy_lines = (line.replace('\\0', '') for line in in_file)\n reader = csv.DictReader(lazy_lines, lineterminator=csvlt, delimiter=csvdel, quotechar=csvquo)\n for row in reader:\n writer.writerow({'COUNTRY': row['COUNTRY'] + '{{ alias }}', 'CARS': row['CARS'] + '{{ size }}'})\nfrom pathlib import Path\nimport sys\ncontents = Path('/data/config.json').read_text()\nprint(contents, file=sys.stdout)" + ] + } + ] + } + ] + }, + "variables_id": "807943784", + "variables_values_id": "807952812", + "image_parameters": {}, + "action": "run", + "authorization": {} +} +``` +
+ +#### Option 2 -- Run a job with stored values +Similarly to the [default values](http://localhost:4000/integrate/variables/#step-2--create-default-values-for-variable), +you can store another set of values. Let's add another configuration row to the *existing* variable configuration: + +```json +{ + "values": [ + { + "name": "alias", + "value": "WATMAN" + }, + { + "name": "size", + "value": "4200" + } + ] +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#fbe487b5-cd68-4318-8219-7c067ebef795). +You will obtain an ID of the row. Then create a table called **watman** with +columns **COUNTRY** and **CARS**. You can use this [sample CSV file](/integrate/variables/countries.csv). + +Run a job with parameters and provide the ID of the main configuration in the `config` property and +the ID of the value row in `variableValuesId`: + +```json +{ + "component": "keboola.python-transformation-v2", + "config": "807968875, + "mode": "run", + "variableValuesId": "807957572" +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#f883eb13-3f20-4e03-bf1b-36e9c889f773). +The output table now contains: + +|COUNTRY|CARS| +|---|---| +|BelgiumWATMAN|62937814200| +|FinlandWATMAN|33582324200| +|ItalyWATMAN|413938774200| + +#### Option 3 -- Run a job with inline values +The last option to provide the values for variables is to enter them directly when running a job. +Variable values are entered in the `variableValuesData` property: + +```json +{ + "component": "keboola.python-transformation-v2", + "config": "807968875", + "mode": "run", + "variableValuesData": { + "values": [ + { + "name": "alias", + "value": "batman" + }, + { + "name": "size", + "value": "scatman" + } + ] + } +} +``` + +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#2c38d6ca-2eda-4c7e-9888-071fad3d31d8). + +The output table will contain: + +|COUNTRY|CARS| +|---|---| +|Belgiumbatman|6293781scatman| +|Finlandbatman|3358232scatman| +|Italybatman|41393877scatman| + +## Orchestrator Integration +Variables in a configuration interact with an orchestrator in two ways: + +- Variables can be entered in task configuration. +- Variables can be entered when running an orchestration. + +Entering variable values in task configurations allows the orchestration to run configurations with variables. +Variable values are entered in the `actionParameters` property. The parameters are identical to +[running a job](/integrate/variables/#step-4--run-job). + +When running an orchestration, you can also provide variable values for an entire orchestration. In that case, +the provided values will override those set in individual orchestration tasks. The parameters are identical +to [running a job](/integrate/variables/#step-4--run-job). + +### Step 5 -- Create Orchestration +You have to use the +[Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) +to create a configuration of the `keboola.orchestrator` component. +You can use the following data in the configuration: + +```json +{ + "phases": [ + { + "id": 2468, + "name": "Extractors", + "dependsOn": [] + } + ], + "tasks": [ + { + "id": 13579, + "name": "Example", + "phase": 2468, + "task": { + "componentId": "keboola.python-transformation-v2", + "configId": "807968875", + "mode": "run", + "variableValuesId": "807952812" + }, + "continueOnFailure": false, + "enabled": true + } + ] +} +``` + +The contents of the `task` property are identical to the body +of the [run job API call](/integrate/variables/#step-4--run-job). Here, the value `807968875` refers to the ID +of the main configuration, and `807952812` refers to the ID of the configuration row with variable values. +You can use the `variableValuesData` field in the same manner. +Creating the above configuration will return a response containing the configuration ID, e.g., `807969959`. +See an [example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#9f2f9da0-59eb-4f33-a206-e5add24725d1). + +### Step 6 -- Run Orchestration +When running an orchestration which contains configurations referencing variables, you have to provide their +values. You can either rely on the stored values (either at the component configuration or in the orchestration task) +or you can provide the values at runtime. + +#### Option 1 -- Rely on stored values +Use the [Run Job API call](https://api.keboola.com/?service=job-queue#job-queue/tag/jobs/POST/jobs) +to run an orchestration. In its simplest form, the request body needs to contain just the ID of the orchestration +(obtained in the previous step): + +```json +{ + "component": "keboola.orchestrator", + "config": "807969959", + "mode": "run" +} +``` + +As long as the variable values can be found somewhere, this is sufficient. See [an example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#3ebdc3f5-a940-4f0d-860b-ec311f704a7e). + +#### Option 2 -- Provide values +Use the [Run Job API call](https://api.keboola.com/?service=job-queue#job-queue/tag/jobs/POST/jobs) +to run an orchestration. Additionally, you can use the `variableValuesId` or `variableValuesData` property +to override variable values set to individual tasks. The calling convention is the same as shown in the +[basic job run](/integrate/variables/#step-4--run-job). The same rules also apply, notably that you can't +use `variableValuesId` and `variableValuesData` together. +A sample request body: + +```json +{ + "component": "keboola.orchestrator", + "config": "807969959", + "mode": "run", + "variableValuesData": { + "values": [ + { + "name": "alias", + "value": "batman" + }, + { + "name": "size", + "value": "scatman" + } + ] + } +} +``` + +See [an example](https://documenter.getpostman.com/view/3086797/77h845D?version=latest#f4fcf7af-afbe-4c29-999e-0f4c50aa477b). + +## Variables Evaluation Sequence +There is a number of places where variable values can be provided (either as a reference to an existing row with +values or as an array of `values`): + +- Parameters in the orchestration +- Parameters in `task` setting of the orchestration +- Parameters in the component job itself +- Default values stored in configuration (`variables_values_id` property) + +The following diagram shows the parameters mentioned on this page and to what they refer to: + +![Screenshot -- Properties references](/integrate/variables/variables.svg) + +In a nutshell, `variableValuesId` always refers to the row of the variable configuration associated with the +main configuration. The main configuration is referenced in the `config` parameter. From another point of view, +the `config` parameter represents the configuration (either a component or an orchestration) to be run. +Note that in stored configurations snake_case is used instead of camelCase. + +The following rules describe the evaluation sequence: + +- Values provided in job parameters (a component job or an orchestration job) override the stored values. +- Values provided in an orchestration job override the stored values in `task`. +- Values provided in `task` override values stored in the component configuration. +- `variableValuesData` and `variableValuesId` can't be used together, so neither of them takes precedence. A reference to stored values can't be mixed with providing the values inline. +- If no values are provided anywhere, the default values are used. If no default values are present, an error is raised. + +## Shared Code +Related to variables is the Shared Code feature. Shared code allows to share parts of configuration code. In a +configuration it is also replaced using the [Moustache syntax](https://mustache.github.io/mustache.5.html). Shared code +is referenced using `shared_code_id` and `shared_code_row_ids` configuration nodes. Unlike variables, shared code can't +be overridden at runtime (so there are no parameters to set when running a job or in orchestration). +Shared code can, however, contain its own variables which need to be merged to those of the main configuration. + +### Creating Shared Code +Shared code pieces is stored as configuration rows of a dedicated component `keboola.shared-code`. Before creating a +piece of a shared code, you first have to create a configuration. Notice that the UI uses certain configurations for +certain components so you might want to check the existing configurations of `keboola.shared-code` component before +crating a new configuration. + +To create a configuration, use the [create configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs). The configuration content is ignored, i.e all you need to provide is name: + +```bash +curl --location --request POST 'https://connection.keboola.com/v2/storage/components/keboola.shared-code/configs' \ +--header 'X-StorageAPI-Token: my-token' \ +--header 'Content-Type: application/x-www-form-urlencoded' \ +--data-urlencode 'name=python-code' +``` + +Let's assume that the created configuration ID is `618884794`. +Next step is to create the shared code piece itself. To do this create a configuration row of the above configuration +with the configuration row content containing a piece of share code, for example: + +```json +{ + "code_content": [ + "from os import listdir\nfrom os.path import isfile, join\n\nmypath = '\''/data/in/files'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)\nmypath = '\''/data/in/user'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)" + ] +} +``` + +It is advisable to set a reasonable `rowId` of the row, because it will be used later to reference the shared code: + +```bash +curl --location --request POST 'https://connection.keboola.com/v2/storage/components/keboola.shared-code/configs/618884794/rows' \ +--header 'X-StorageApi-Token: my-token' \ +--header 'Content-Type: application/x-www-form-urlencoded' \ +--data-urlencode 'configuration={ + "code_content": ["from os import listdir\nfrom os.path import isfile, join\n\nmypath = '\''/data/in/files'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)\nmypath = '\''/data/in/user'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)"] +} +' \ +--data-urlencode 'rowId=dumpfiles' +``` + +The above example creates a piece of shared python code named `dumpfiles` which contains the +following python code: + +```python +from os import listdir +from os.path import isfile, join + +mypath = '/data/in/files' +onlyfiles = [f for f in listdir(mypath)] +print(onlyfiles) +mypath = '/data/in/user' +onlyfiles = [f for f in listdir(mypath)] +print(onlyfiles) +``` + +### Using Shared Code +To use a piece of shared code, you have to reference it in a configuration using `shared_code_id` which is the ID of the shared code configuration and `shared_code_row_ids` which is an array of IDS of shared code pieces. With the above example you need to add the following nodes to the configuration: + +```json +{ + "storage": {...}, + "parameters": {...}, + "shared_code_id": "618884794", + "shared_code_row_ids": ["dumpfiles"] +} +``` + +With that all moustache references to `{{ dumpfiles}}` will be replaced by the shared code piece. All other +moustache references will be kept untouched and be treated like variables. E.g: the following configuration: + +```json +{ + "storage": {}, + "parameters": { + "blocks": [ + { + "name": "Main block", + "codes": [ + { + "name": "Main code", + "script": ["{{ someOtherPlaceholder}}"] + }, + { + "name": "Debug", + "script": ["{{ dumpfiles}}"] + } + ] + } + ] + }, + "variables_id": "618878103", + "variables_values_id": "618878104", + "shared_code_id": "618884794", + "shared_code_row_ids": ["dumpfiles"] +} +``` + +Will be modified to: + +```json +{ + "storage": {}, + "parameters": { + "blocks": [ + { + "name": "Main block", + "codes": [ + { + "name": "Main code", + "script": ["{{ someOtherPlaceholder}}"] + }, + { + "name": "Debug", + "script": ["from os import listdir\nfrom os.path import isfile, join\n\nmypath = '\''/data/in/files'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)\nmypath = '\''/data/in/user'\''\nonlyfiles = [f for f in listdir(mypath)]\nprint(onlyfiles)"] + } + ] + } + ] + }, + "variables_id": "618878103", + "variables_values_id": "618878104", + "shared_code_id": "618884794", + "shared_code_row_ids": ["dumpfiles"] +} +``` + +The variables then need to contain `someOtherPlaceholder` variable in order to produce a fully functional configuration. +The same way if the shared code piece contains any variables, they have to be set when running the configuration. + +**Important:** The replacement of the shared code piece occurs only within an array of the configuration JSON. In the above code, the shared code reference is `"script": ["{{ someOtherPlaceholder}}"]` which is the only valid form of a Shared Code reference. For example +`"script": ["some code {{ someOtherPlaceholder}} some other code"]` or `"script": "{{ someOtherPlaceholder}}"` are invalid Shared Code references which may not be replaced the way you intend. + +**Important:** The replacement of the shared code piece merges the `code_content` array containing the shared code definition with the array containing the shared code reference. With a shared code reference in form `"script": ["a", "{{ someOtherPlaceholder}}", "b"]` and shared code definition in form `"code_content": ["c", "d"]` the resulting replacement would be `"script": ["a", "c", "d", "b"]`. diff --git a/src/content/docs/integrate/variables/tutorial-1.png b/src/content/docs/integrate/variables/tutorial-1.png new file mode 100644 index 000000000..736929496 Binary files /dev/null and b/src/content/docs/integrate/variables/tutorial-1.png differ diff --git a/src/content/docs/integrate/variables/tutorial-2.png b/src/content/docs/integrate/variables/tutorial-2.png new file mode 100644 index 000000000..d0730e343 Binary files /dev/null and b/src/content/docs/integrate/variables/tutorial-2.png differ diff --git a/src/content/docs/integrate/variables/tutorial/index.md b/src/content/docs/integrate/variables/tutorial/index.md new file mode 100644 index 000000000..9a025d937 --- /dev/null +++ b/src/content/docs/integrate/variables/tutorial/index.md @@ -0,0 +1,197 @@ +--- +title: Variables Tutorial +slug: 'integrate/variables/tutorial' +--- + + +This tutorial will guide you through basic usage of [Variables](/integrate/variables/) in the component configuration. +The result will be the parametrized configuration of the [Generic Extractor](/extend/generic-extractor), +but this approach can be applied to any component. + +In the examples, we use the `curl` console tool to interact with our APIs. + +## Define API endpoints + +First, store the [API endpoints](/overview/api/) as environment variables, so we don't have to repeat ourselves. + +We will need: +- [Storage API](/integrate/storage/api/) to store the variable definitions and the extractor configuration - +- [Job Queue API](/extend/job-queue/) to run the extractor job from the configuration. + +The host names depend on your [stack](/overview/api/#stacks-and-endpoints): + +```shell +export STORAGE_API_HOST="https://connection.keboola.com" +export JOB_QUEUE_HOST="https://queue.keboola.com" +``` + +## Obtain Storage API Token + +A [Storage API Token](/management/project/tokens/) is needed to interact with the [Keboola APIs](/overview/api/#list-of-keboola-apis). + +Obtain a Storage API token from the user interface of your project, see this [Guide](/management/project/tokens). + +Then store the token to the environment variable. +```shell +export TOKEN="..." +``` + +## Define variables + +The next step is to define the variables in a [Variable Configuration](/integrate/variables/#variable-configuration). + +Define name and type of the variables. +```shell +export VARIABLE_CONFIG_NAME="Extractor variables" +export VARIABLE_CONFIG=' +{ + "variables": [ + { + "name": "outputBucket", + "type": "string" + }, + { + "name": "id", + "type": "int" + } + ] +} +' +``` + +Use [Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) to store *variable configuration*. +```shell +curl --include \ + --request POST \ + --header "Content-Type: application/x-www-form-urlencoded" \ + --header "X-StorageApi-Token: $TOKEN" \ + --data-urlencode "name=$VARIABLE_CONFIG_NAME" \ + --data-urlencode "configuration=$VARIABLE_CONFIG" \ +"$STORAGE_API_HOST/v2/storage/components/keboola.variables/configs" +``` + +Example API call result. +```json +{ + "id":"1234", + "name":"Extractor variables", + "description":"..." +} +``` + +Save *variable configuration* `id` from the the result to the environment variable. +```shell +export VARIABLE_CONFIG_ID="1234" +``` + +**The created *variable configuration* defines the names and types of variables.** + +You can create additional configurations that contain (default) [Variable Values](/integrate/variables/#variable-values). + +In this example, the values of the variables are entered directly to the [run API call](#run-extractor-configuration) (see bellow), +so configuration with the variable values is not used. + +## Create extractor configuration + +Define *extractor configuration* with variables `{{placeholders}}`. + +```shell + +export COMPONENT_ID="ex-generic-v2" +export EXTRACTOR_CONFIG_NAME="Extractor configuration" +export EXTRACTOR_CONFIG=' +{ + "parameters": { + "api": { + "baseUrl": "https://jsonplaceholder.typicode.com/" + }, + "config": { + "debug": true, + "outputBucket": "{{outputBucket}}", + "jobs": [ + { + "endpoint": "posts/{{id}}/comments" + } + ] + } + }, + "variables_id": "'$VARIABLE_CONFIG_ID'" +} +' + +``` + +Use [Create Configuration API call](https://api.keboola.com/?service=storage#post-/v2/storage/branch/-branchId-/components/-componentId-/configs) to store extractor configuration. +```shell +curl --include \ + --request POST \ + --header "Content-Type: application/x-www-form-urlencoded" \ + --header "X-StorageApi-Token: $TOKEN" \ + --data-urlencode "name=$EXTRACTOR_CONFIG_NAME" \ + --data-urlencode "configuration=$EXTRACTOR_CONFIG" \ +"$STORAGE_API_HOST/v2/storage/components/$COMPONENT_ID/configs" +``` + +Example API call result. +```json +{ + "id":"4567", + "name":"Extractor configuration", + "description":"..." +} +``` + +Save *extractor configuration* `id` from the result to the environment variable. +```shell +export EXTRACTOR_CONFIG_ID="4567" +``` + +## Run extractor configuration + +Define values of the variables. +```shell +export VARIABLES_VALUES=' +[ + {"name": "outputBucket", "value": "my-bucket"}, + {"name": "id", "value": 1} +] +' +``` + +In this example are values of the variables part of the run job request. + +For other ways to define values see the [Variables documentation](/integrate/variables/#variable-values). + +Use [Run Job API call](https://api.keboola.com/?service=job-queue#post-/jobs) to run *extractor configuration*. +```shell +curl --include \ + --request POST \ + --header "Content-Type: application/json" \ + --header "X-StorageApi-Token: $TOKEN" \ + --data-binary ' + { + "component": "'$COMPONENT_ID'", + "config": "'$EXTRACTOR_CONFIG_ID'", + "mode": "run", + "variableValuesData": { + "values": '$VARIABLES_VALUES' + } + } + ' \ +"$JOB_QUEUE_HOST/jobs" +``` + +## Check the job result + +The status of a running job can be seen via API or UI. + +In the picture we can see that the entered values of the variables were used. + +![Screenshot -- Job](/integrate/variables/tutorial-1.png) + +A note about the replaced variables is in the job logs. + +![Screenshot -- Job Logs](/integrate/variables/tutorial-2.png) + +See the [Variables documentation](/integrate/variables/#variable-values) for more information. + diff --git a/src/content/docs/integrate/variables/variables.svg b/src/content/docs/integrate/variables/variables.svg new file mode 100644 index 000000000..5433c2a1c --- /dev/null +++ b/src/content/docs/integrate/variables/variables.svg @@ -0,0 +1,3 @@ + + +
variables_id
variables_id
variable_values_id
variable_values_id
Main Configuration
(vendor.component)
Main Configuration...
Variables Configuration
(keboola.variables)
Variables Configurat...
config
config
variableValuesId
variableValuesId
Orchestration
Orchestration
Variable Values Row
Variable Values...
config
config
variableValuesId
variableValuesId
Run Orchestration
Run O...
config
config
variableValuesId
variableValuesId
Run Configuration
Run C...
Viewer does not support full SVG 1.1
\ No newline at end of file diff --git a/src/content/docs/kbc_structure.png b/src/content/docs/kbc_structure.png new file mode 100644 index 000000000..3f859daba Binary files /dev/null and b/src/content/docs/kbc_structure.png differ diff --git a/src/content/docs/management/jobs/index.md b/src/content/docs/management/jobs/index.md index bdddd82d5..108cb091e 100644 --- a/src/content/docs/management/jobs/index.md +++ b/src/content/docs/management/jobs/index.md @@ -16,7 +16,7 @@ All jobs are logged and their tracked history is virtually unlimited. Click on a - what tables were exported (read from your Storage by the job). - how many [credits](/management/project/limits/#project-power) were used by running the job. - what events occurred during the job execution. -- what exact parameters were used for the job (this might be useful when working with the [API](https://developers.keboola.com/integrate/jobs/#apis-for-working-with-jobs)). +- what exact parameters were used for the job (this might be useful when working with the [API](/integrate/jobs/#apis-for-working-with-jobs)). ![Screenshot - Jobs Detail](/management/jobs/jobs-detail.png) @@ -57,7 +57,7 @@ Using the search box and advanced patterns you can easily find job based on vari | **All non-successful jobs from either HTTP or Google Sheets writer** | `params.component:(keboola.ex-http OR keboola.wr-google-sheets) AND -status:success` | For more technical information about background jobs, see our -[Developers documentation](https://developers.keboola.com/integrate/jobs/). +[Developers documentation](/integrate/jobs/). ## Running Jobs Jobs are either run [manually from any configuration](/tutorial/) or automatically by the @@ -80,7 +80,7 @@ Terminating the child job will probably cause the parent to terminate or fail. ## Waiting Jobs When a job is run, it is always put in the waiting state to wait for our **infrastructure** --- -[worker](https://developers.keboola.com/integrate/jobs/) to start executing it. +[worker](/integrate/jobs/) to start executing it. This usually takes anywhere from several seconds to a couple of minutes at most. There is one more reason for a job to be in the waiting state: **project parallelism limits**. diff --git a/src/content/docs/management/project/index.md b/src/content/docs/management/project/index.md index 4041c2a70..189aed10d 100644 --- a/src/content/docs/management/project/index.md +++ b/src/content/docs/management/project/index.md @@ -35,7 +35,7 @@ The API Tokens section in the Keboola platform is used to manage programmatic ac ### 4 CLI Sync The CLI Sync section is used to set up and manage synchronization between the Keboola platform and your local development environment using the Keboola CLI (KBC CLI). It allows you to securely connect CLI-based tools to a specific project, enabling actions like pulling configurations, pushing changes, or running jobs programmatically. -If you need to setup your Keboola CLI, simply follow the instructions displayed on this page. For more detailed information go to the [Developer documentation](https://developers.keboola.com/cli/). +If you need to setup your Keboola CLI, simply follow the instructions displayed on this page. For more detailed information go to the [Developer documentation](/cli/keboola-as-code/). ### 5 Features The Features section in the Keboola UI is used to toggle project-specific feature flags. It allows project admins to enable or disable experimental, beta, or advanced platform capabilities that are not generally available by default. diff --git a/src/content/docs/management/project/tokens/index.md b/src/content/docs/management/project/tokens/index.md index 4bbe743f1..8f55d57d7 100644 --- a/src/content/docs/management/project/tokens/index.md +++ b/src/content/docs/management/project/tokens/index.md @@ -21,7 +21,7 @@ Normally, when you are using the user interface, your API token is exchanged aut the server backend. Therefore you need to work with tokens only when working with Keboola programmatically (or if you need to limit a user's authorization to certain operations or data). To learn more about all the available programmatic approaches, please follow our -[developers documentation](https://developers.keboola.com/overview/api/). +[developers documentation](/overview/api/). Tokens can be managed from the **Project Settings > API Tokens** page. @@ -49,7 +49,7 @@ API tokens are created Automatically created tokens have the lowest possible permissions for their task and also set expiration if possible. These are the typical reasons to manually create a new API token: -- You want to use the [APIs](https://developers.keboola.com/overview/api/); this includes all of the [Storage clients](https://developers.keboola.com/integrate/storage/#storage-api-clients). +- You want to use the [APIs](/overview/api/); this includes all of the [Storage clients](/integrate/storage/#storage-api-clients). - You need to limit access to certain data (for example, share a single table) or components. Although tokens cannot be used to directly log in to the Keboola user interface, they do allow executing almost all @@ -61,7 +61,7 @@ token string was revealed to unauthorized persons. When creating a new token, the following rules apply: - Tokens by default give **no access** to any of the Keboola component configurations. -- Token bearers can only access **permitted** Storage buckets via the [Storage API](http://developers.keboola.com/integrate/storage/) or +- Token bearers can only access **permitted** Storage buckets via the [Storage API](/integrate/storage/) or [Storage console](https://storage-api-console.keboola.com/). - Tokens **cannot** be used to run any actions in your project. However, they can trigger flows. - Tokens **cannot** be used to create other tokens (only a master token can be used to create new tokens). @@ -162,8 +162,8 @@ your data with selected users, the buckets can be also used for writing; people can send data directly to your Keboola project instead of struggling with FTP or e-mail attachments. To revoke the access, simply delete or refresh the token. -The token can then be used with the [Storage API](https://developers.keboola.com/integrate/) -or [other APIs](https://developers.keboola.com/overview/api/). +The token can then be used with the [Storage API](/integrate/) +or [other APIs](/overview/api/). ### Storage Console Typical usecase of sharing a token with someone is giving them a partial access to your project storage. The @@ -177,7 +177,7 @@ The Storage Console allows some basic operations with the project [Storage](/sto ![Screenshot - Storage Console](/management/project/tokens/storage-console.png) The link to the Storage API Console is available at the token retrieval page as it is different for each -[region](https://developers.keboola.com/overview/api/): +[region](/overview/api/): - [AWS US Region](https://storage-api-console.keboola.com/?endpoint=https%3A%2F%2Fconnection.keboola.com) - [AWS EU Region](https://storage-api-console.keboola.com/?endpoint=https%3A%2F%2Fconnection.eu-central-1.keboola.com) diff --git a/src/content/docs/overview/api/apiary-console.png b/src/content/docs/overview/api/apiary-console.png new file mode 100644 index 000000000..1965d06e3 Binary files /dev/null and b/src/content/docs/overview/api/apiary-console.png differ diff --git a/src/content/docs/overview/api/index.md b/src/content/docs/overview/api/index.md new file mode 100644 index 000000000..f68278c53 --- /dev/null +++ b/src/content/docs/overview/api/index.md @@ -0,0 +1,276 @@ +--- +title: Our APIs +slug: 'overview/api' +--- + + +All our [Keboola services](/overview/) have a public API on [api.keboola.com](https://api.keboola.com/). We recommend using either the API Console or Postman Client for sending requests to our +API. Most of our APIs accept and return data in JSON format. +Many of these APIs require a *Storage API token*, specified in the `X-StorageApi-Token` header. + +## List of Keboola APIs + +All parts of the Keboola platform can be controlled via an API. +The main APIs for our components are: + +
+Note: The api.keboola.com links in the table below open the API documentation portal for the US Virginia AWS stack. +If you are using a different stack, navigate to your stack's API portal first — see API Documentation Portals below — and then select the service there. +Using a portal for a different stack than your token's stack will result in Invalid Token errors. +
+ +| API | Description | +|-------------------------------------------------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| [Keboola Storage API](https://api.keboola.com/?service=storage) ([source](https://github.com/keboola/storage-api-php-client/blob/master/apiary.apib)) | [Storage](/integrate/storage/) is the main Keboola component storing all data. | +| [Keboola Management API](https://api.keboola.com/?service=manage) | API managing Keboola projects and users (and notifications and features). | +| [AI API](https://api.keboola.com/?service=ai) | API for supporting AI features. | +| [Billing API](https://api.keboola.com/?service=billing) | Billing API for Pay as You Go projects. | +| [Developer Portal API](https://api.keboola.com/?service=developer-portal) | Developer Portal is an application separated from Keboola for [creating components](/extend/component/). | +| [Editor API](https://api.keboola.com/?service=editor) | API for managing SQL editor sessions. | +| [Encryption API](https://api.keboola.com/?service=encryption) | Provides [Encryption](/overview/encryption/). | +| [Importer API](https://api.keboola.com/?service=import) | [Importer](/integrate/storage/api/importer/) is a helper service for easy table imports. | +| [Notifications API](https://api.keboola.com/?service=notification) | API to subscribe to events, e.g., failed orchestrations. | +| [OAuth Broker API](https://api.keboola.com/?service=oauth) | OAuth Broker is a component managing [OAuth authorizations](/extend/common-interface/oauth/#authorize) of other components. | +| [Query API](https://api.keboola.com/?service=query) | Query is a service for running SQL queries on Snowflake and BigQuery. | +| [Queue API](https://api.keboola.com/?service=job-queue) | Queue is a service for [running components](/extend/job-queue/) and managing [Jobs](/integrate/jobs/). | +| [Sandboxes Service API](https://api.keboola.com/?service=sandboxes-service) | API for managing Apps and Python/R workspaces. | +| [Scheduler API](https://api.keboola.com/?service=scheduler) | API to automate configurations. | +| [Stream API](https://api.keboola.com/?service=stream) | The Keboola Stream API allows you to ingest small and frequent events into your project's storage. | +| [Synchronous Actions API](https://api.keboola.com/?service=sync-actions) | API to trigger [Synchronous Actions](/extend/common-interface/actions/). | +| [Vault](https://api.keboola.com/?service=vault) | Service handling variables & credentials storage. | + +If you're unsure which API to use, refer to our [integration guide](/integrate/). It describes the roles of different APIs and contains examples of commonly +performed actions. + +## Stacks and Endpoints +Keboola is available in multiple [stacks](/overview/#stacks), which can be +either multi-tenant or single-tenant. Current multi-tenant stacks are: + +- US Virginia AWS – [connection.keboola.com](https://connection.keboola.com/) +- US Virginia GCP - [connection.us-east4.gcp.keboola.com](https://connection.us-east4.gcp.keboola.com/) +- EU Frankfurt AWS – [connection.eu-central-1.keboola.com](https://connection.eu-central-1.keboola.com/) +- EU Ireland Azure – [connection.north-europe.azure.keboola.com](https://connection.north-europe.azure.keboola.com/) +- EU Frankfurt GCP - [connection.europe-west3.gcp.keboola.com](https://connection.europe-west3.gcp.keboola.com/) + +Each stack operates as an independent instance of Keboola services with its own data, users, and tokens. +Single-tenant stacks are available for a single enterprise customer, with a domain name +in the format `connection.CUSTOMER_NAME.keboola.com`. + +### API Documentation Portals + +The API documentation portal (`api.*`) is deployed independently per stack. Always use the portal +for your own stack — tokens are not valid across stacks, and using the wrong portal will cause +`Invalid Token` errors when trying out API calls. + +| Stack | API Documentation Portal | +|---|---| +| US Virginia AWS | [api.keboola.com](https://api.keboola.com/) | +| EU Frankfurt AWS | [api.eu-central-1.keboola.com](https://api.eu-central-1.keboola.com/) | +| EU Ireland Azure | [api.north-europe.azure.keboola.com](https://api.north-europe.azure.keboola.com/) | +| EU Frankfurt GCP | [api.europe-west3.gcp.keboola.com](https://api.europe-west3.gcp.keboola.com/) | +| US Virginia GCP | [api.us-east4.gcp.keboola.com](https://api.us-east4.gcp.keboola.com/) | + +### Machine-Readable API Index + +For agentic usage and tooling (AI agents, MCP servers, CI), each stack's API portal also publishes a +machine-readable index of its APIs at `https://api./apis.json` — for example, +[api.keboola.com/apis.json](https://api.keboola.com/apis.json). The index lists each available service with +its base `apiUrl` and a link to its OpenAPI specification (`openApiSpecUrl`), so tools can discover and load +the specs programmatically: + +```json +{ + "stack": "keboola.com", + "services": [ + { + "id": "storage", + "name": "Storage API", + "apiUrl": "https://connection.keboola.com", + "openApiSpecUrl": "https://api.keboola.com/specs/storage.json" + } + ] +} +``` + +The index is stack-specific (excluded services are omitted). The raw specs under `/specs/` keep their original +`servers`, so consumers should use the `apiUrl` from the index as the base URL. The `openApiSpecUrl` extension +mirrors the source document (`.json` or `.yaml`) — use the exact URL from the index rather than assuming one. + +### Service Endpoints + +If you are calling the APIs directly (not through the portal), modify the hostname accordingly. +Otherwise, you may encounter `Invalid Token` or unauthorized errors. The *authoritative list* of available endpoints is provided by the [Storage API Index Call](https://api.keboola.com/?service=storage#get-/v2/storage/branch/-branchId-/components/-componentId-). The following is a sample response: + +```json +{ + ..., + "services": [ + { + "id": "import", + "url": "https://import.keboola.com" + }, + { + "id": "oauth", + "url": "https://oauth.keboola.com" + }, + { + "id": "queue", + "url": "https://queue.keboola.com" + }, + { + "id": "billing", + "url": "https://billing.keboola.com" + }, + { + "id": "encryption", + "url": "https://encryption.keboola.com" + }, + { + "id": "scheduler", + "url": "https://scheduler.keboola.com" + }, + { + "id": "sync-actions", + "url": "https://sync-actions.keboola.com" + }, + { + "id": "notification", + "url": "https://notification.keboola.com" + } + ], +} +``` + +The services listed above are: + +- `import` --- [Storage Importer Service](/integrate/storage/api/importer/) +- `oauth` --- [OAuth Manager Service](/extend/common-interface/oauth/) +- `queue` --- [Service for Running Components](/extend/job-queue/) +- `billing` --- Service for Computing Credits +- `encryption` --- Service for [Encryption](/overview/encryption/) +- `scheduler` --- [Service for Configuring Schedules](/automate/set-schedule/) +- `sync-actions` --- [Service for Running Synchronous Actions](/extend/common-interface/actions/) +- `notification` --- Service for Configuring Job Notifications + +For convenience, the following table lists active services and their URLs, though for an authoritative answer +and in application integrations, we strongly suggest using the above API call. + +| API | Service | Region | URL | +|------------------------|----------------|------------------|-----------------------------------------------------| +| AI | `ai` | US Virginia AWS | https://ai.keboola.com | +| AI | `ai` | US Virginia GCP | https://ai.us-east4.gcp.keboola.com | +| AI | `ai` | EU Frankfurt AWS | https://ai.eu-central-1.keboola.com | +| AI | `ai` | EU Ireland Azure | https://ai.north-europe.azure.keboola.com | +| AI | `ai` | EU Frankfurt GCP | https://ai.europe-west3.gcp.keboola.com | +| Billing | `billing` | US Virginia AWS | https://billing.keboola.com | +| Billing | `billing` | US Virginia GCP | https://billing.us-east4.gcp.keboola.com | +| Billing | `billing` | EU Frankfurt AWS | https://billing.eu-central-1.keboola.com | +| Billing | `billing` | EU Ireland Azure | https://billing.north-europe.azure.keboola.com | +| Billing | `billing` | EU Frankfurt GCP | https://billing.europe-west3.gcp.keboola.com | +| Developer Portal | `developer` | US Virginia AWS | https://developer.keboola.com | +| Developer Portal | `developer` | US Virginia GCP | https://developer.us-east4.gcp.keboola.com | +| Developer Portal | `developer` | EU Frankfurt AWS | https://developer.eu-central-1.keboola.com | +| Developer Portal | `developer` | EU Ireland Azure | https://developer.north-europe.azure.keboola.com | +| Developer Portal | `developer` | EU Frankfurt GCP | https://developer.europe-west3.gcp.keboola.com | +| Editor | `editor` | US Virginia AWS | https://editor.keboola.com | +| Editor | `editor` | US Virginia GCP | https://editor.us-east4.gcp.keboola.com | +| Editor | `editor` | EU Frankfurt AWS | https://editor.eu-central-1.keboola.com | +| Editor | `editor` | EU Ireland Azure | https://editor.north-europe.azure.keboola.com | +| Editor | `editor` | EU Frankfurt GCP | https://editor.europe-west3.gcp.keboola.com | +| Encryption | `encryption` | US Virginia AWS | https://encryption.keboola.com | +| Encryption | `encryption` | US Virginia GCP | https://encryption.us-east4.gcp.keboola.com | +| Encryption | `encryption` | EU Frankfurt AWS | https://encryption.eu-central-1.keboola.com | +| Encryption | `encryption` | EU Ireland Azure | https://encryption.north-europe.azure.keboola.com | +| Encryption | `encryption` | EU Frankfurt GCP | https://encryption.europe-west3.gcp.keboola.com | +| Importer | `import` | US Virginia AWS | https://import.keboola.com | +| Importer | `import` | US Virginia GCP | https://import.us-east4.gcp.keboola.com | +| Importer | `import` | EU Frankfurt AWS | https://import.eu-central-1.keboola.com | +| Importer | `import` | EU Ireland Azure | https://import.north-europe.azure.keboola.com | +| Importer | `import` | EU Frankfurt GCP | https://import.europe-west3.gcp.keboola.com | +| Management | `management` | US Virginia AWS | https://management.keboola.com | +| Management | `management` | US Virginia GCP | https://management.us-east4.gcp.keboola.com | +| Management | `management` | EU Frankfurt AWS | https://management.eu-central-1.keboola.com | +| Management | `management` | EU Ireland Azure | https://management.north-europe.azure.keboola.com | +| Management | `management` | EU Frankfurt GCP | https://management.europe-west3.gcp.keboola.com | +| Notification | `notification` | US Virginia AWS | https://notification.keboola.com | +| Notification | `notification` | US Virginia GCP | https://notification.us-east4.gcp.keboola.com | +| Notification | `notification` | EU Frankfurt AWS | https://notification.eu-central-1.keboola.com | +| Notification | `notification` | EU Ireland Azure | https://notification.north-europe.azure.keboola.com | +| Notification | `notification` | EU Frankfurt GCP | https://notification.europe-west3.gcp.keboola.com | +| OAuth | `oauth` | US Virginia AWS | https://oauth.keboola.com | +| OAuth | `oauth` | US Virginia GCP | https://oauth.europe-west3.gcp.keboola.com | +| OAuth | `oauth` | EU Frankfurt AWS | https://oauth.eu-central-1.keboola.com | +| OAuth | `oauth` | EU Ireland Azure | https://oauth.north-europe.azure.keboola.com | +| OAuth | `oauth` | EU Frankfurt GCP | https://oauth.europe-west3.gcp.keboola.com | +| Query | `query` | US Virginia AWS | https://query.keboola.com | +| Query | `query` | US Virginia GCP | https://query.us-east4.gcp.keboola.com | +| Query | `query` | EU Frankfurt AWS | https://query.eu-central-1.keboola.com | +| Query | `query` | EU Ireland Azure | https://query.north-europe.azure.keboola.com | +| Query | `query` | EU Frankfurt GCP | https://query.europe-west3.gcp.keboola.com | +| Queue | `queue` | US Virginia AWS | https://queue.keboola.com | +| Queue | `queue` | US Virginia GCP | https://queue.us-east4.gcp.keboola.com | +| Queue | `queue` | EU Frankfurt AWS | https://queue.eu-central-1.keboola.com | +| Queue | `queue` | EU Ireland Azure | https://queue.north-europe.azure.keboola.com | +| Queue | `queue` | EU Frankfurt GCP | https://queue.europe-west3.gcp.keboola.com | +| Scheduler | `scheduler` | US Virginia AWS | https://scheduler.keboola.com | +| Scheduler | `scheduler` | US Virginia GCP | https://scheduler.us-east4.gcp.keboola.com | +| Scheduler | `scheduler` | EU Frankfurt AWS | https://scheduler.eu-central-1.keboola.com | +| Scheduler | `scheduler` | EU Ireland Azure | https://scheduler.north-europe.azure.keboola.com | +| Scheduler | `scheduler` | EU Frankfurt GCP | https://scheduler.europe-west3.gcp.keboola.com | +| Storage | | US Virginia AWS | https://connection.keboola.com/ | +| Storage | | US Virginia GCP | https://connection.us-east4.gcp.keboola.com | +| Storage | | EU Frankfurt AWS | https://connection.eu-central-1.keboola.com/ | +| Storage | | EU Ireland Azure | https://connection.north-europe.azure.keboola.com/ | +| Storage | | EU Frankfurt GCP | https://connection.europe-west3.gcp.keboola.com/ | +| Stream | `stream` | US Virginia AWS | https://stream.keboola.com | +| Stream | `stream` | US Virginia GCP | https://stream.us-east4.gcp.keboola.com | +| Stream | `stream` | EU Frankfurt AWS | https://stream.eu-central-1.keboola.com | +| Stream | `stream` | EU Ireland Azure | https://stream.north-europe.azure.keboola.com | +| Stream | `stream` | EU Frankfurt GCP | https://stream.europe-west3.gcp.keboola.com | +| Sync Actions | `sync-actions` | US Virginia AWS | https://sync-actions.keboola.com/ | +| Sync Actions | `sync-actions` | US Virginia GCP | https://sync-actions.us-east4.gcp.keboola.com | +| Sync Actions | `sync-actions` | EU Frankfurt AWS | https://sync-actions.eu-central-1.keboola.com | +| Sync Actions | `sync-actions` | EU Ireland Azure | https://sync-actions.north-europe.azure.keboola.com | +| Sync Actions | `sync-actions` | EU Frankfurt GCP | https://sync-actions.europe-west3.gcp.keboola.com | +| Vault | `vault` | US Virginia AWS | https://vault.keboola.com | +| Vault | `vault` | US Virginia GCP | https://vault.us-east4.gcp.keboola.com | +| Vault | `vault` | EU Frankfurt AWS | https://vault.eu-central-1.keboola.com | +| Vault | `vault` | EU Ireland Azure | https://vault.north-europe.azure.keboola.com | +| Vault | `vault` | EU Frankfurt GCP | https://vault.europe-west3.gcp.keboola.com | + +***Important**: Each stack also uses its own set of [IP addresses](/extractors/ip-addresses/).* + +## Calling API + +There are several ways to send requests to our APIs: + +### Apiary Console +Send requests to our API directly from the Apiary console by clicking on **Switch to console** or **Try**. +Fill in the request headers and parameters, then click **Call Resource**. + +![Apiary console](/overview/api/apiary-console.png) + +The Apiary console is fine if you send API requests only occasionally. It requires no application installation; +however, it has no history and no other useful features. + +### Postman Client +[Postman](https://www.getpostman.com/) is a generic HTTP API client, suitable for more regular API work. +We also provide a collection of [useful API calls](https://documenter.getpostman.com/view/3086797/kbc-samples/77h845D?version=latest#9b9f3e7b-de3b-4c90-bad6-a8760e3852eb) with examples. +The collection contains code examples in various languages; the requests can also be imported into the Postman application. + +![Postman Docs](/overview/api/postman-import.png) + +### cURL +[cURL](https://curl.haxx.se/) is a common library with a [command-line interface (CLI)](https://curl.haxx.se/docs/manpage.html). +You can use the cURL CLI to create simple scripts for interacting with Keboola APIs. For example, to [run a job](/integrate/jobs/): + +```shell +curl --location --request POST 'https://queue.keboola.com/jobs' \ +--header 'X-StorageApi-Token: YourStorageToken' \ +--header 'Content-Type: application/json' \ +--data-raw '{ + "mode": "run", + "component": "keboola.ex-db-mysql", + "config": "sampledatabase" +}' +``` diff --git a/src/content/docs/overview/api/postman-import.png b/src/content/docs/overview/api/postman-import.png new file mode 100644 index 000000000..12333eb04 Binary files /dev/null and b/src/content/docs/overview/api/postman-import.png differ diff --git a/src/content/docs/overview/encryption-1.png b/src/content/docs/overview/encryption-1.png new file mode 100644 index 000000000..9d28c0707 Binary files /dev/null and b/src/content/docs/overview/encryption-1.png differ diff --git a/src/content/docs/overview/encryption-2.png b/src/content/docs/overview/encryption-2.png new file mode 100644 index 000000000..51bb3da41 Binary files /dev/null and b/src/content/docs/overview/encryption-2.png differ diff --git a/src/content/docs/overview/encryption/index.md b/src/content/docs/overview/encryption/index.md new file mode 100644 index 000000000..bd393485c --- /dev/null +++ b/src/content/docs/overview/encryption/index.md @@ -0,0 +1,146 @@ +--- +title: Encryption +slug: 'overview/encryption' +--- + + +Many [Keboola components](/overview/) use the Encryption API to encrypt sensitive values +intended for secure storage. These values are then decrypted within the component itself. +This process ensures that the encrypted values are only accessible inside the components and not +by API users. Additionally, no decryption API is available, meaning end-users cannot decrypt +these values. + +Decryption occurs solely during the serialization of configuration to the Docker container's +configuration file. The decrypted data are stored on the Docker host drive and are promptly +deleted after the container's completion. The component code exclusively accesses the decrypted data. + +## UI Interaction +When saving arbitrary configuration data, if a key is prefixed with the `#` character, the associated value is automatically encrypted. +For instance, consider the following configuration: + +![Screenshot - Configuration editor - before](/overview/encryption-1.png) + +After saving, the configuration appears as follows: + +![Screenshot - Configuration editor - after](/overview/encryption-2.png) + +Once saved, the value becomes encrypted and irreversible. The component defines which values are +encrypted, indicating that not all values can be encrypted unless explicitly supported by the component. + +For example, a component requiring the following configuration: + +```json +{ + "username": "JohnDoe", + "#password": "password" +} +``` + +indicates that the password will be encrypted while the username will not. Adding a +prefix `#` to `username` is ineffective, as the component does not recognize such a key, +even though its value would be encrypted and decrypted normally. Internally, the +[Encryption API](#encrypting-data-with-api) encrypts these values before saving. + +### UI Configuration Adjustment +The UI prioritizes encrypted values over plain ones. If both `password` and `#password` are provided, only `#password` will be retained. +Consequently, this configuration: + +```json +{ + "username": "JohnDoe", + "#password": "KBC::ProjectSecure::ENCODEDSTRING", + "password": "secret", +} +``` + +will be transformed to: + +```json +{ + "username": "JohnDoe", + "#password": "KBC::ProjectSecure::ENCODEDSTRING" +} +``` + +## Encrypting Data with API +The [Encryption API](https://api.keboola.com/?service=encryption#post-/encrypt) can handle +both strings and arbitrary JSON data. For strings, the entire string is encrypted. In JSON data, +only scalar keys starting with `#` are encrypted. For example, encrypting the following: + +```json +{ + "foo": "bar", + "#encryptMe": "secret", + "#encryptMeToo": { + "another": "secret" + } +} +``` + +results in: + +```json +{ + "foo": "bar", + "#encryptMe": "KBC::ProjectSecure::ENCODEDSTRING", + "#encryptMeToo": { + "another": "secret" + } +} +``` + +To encrypt a single string, such as a password, submit the text string for encryption +(no JSON or quotation is used). For example, encrypting + + mySecretPassword + +yields + + KBC::ProjectSecure::ENCODEDSTRING + +The `Content-Type` header in the request differentiates whether the body is treated as a string (`text/plain`) or JSON (`application/json`). + +### Encryption Parameters +The Encryption API accepts the following **optional** parameters: + +- `componentId` --- ID of a [Keboola component](/extend/component/tutorial/#creating-a-component), +- `projectId` --- ID of a Keboola project, +- `configId` --- ID of a component configuration, and +- `branchType` --- Branch type --- either `default` (meaning the default production branch) or `dev` (meaning any development branch other than the production). + +The cipher created depends on the provided parameters: + +- With only `componentId`, the cipher starts with `KBC::ComponentSecure::` and is decryptable +across all configurations of that component. This is recommended for **component-specific secrets** +applicable across all customers (e.g., as a master authorization token). + +- Adding `projectId` to the `componentId` changes the prefix to `KBC::ProjectSecure::`, making the cipher decryptable within +the project's component configurations. This is recommended for **all secrets** used within a typical Keboola project. + +- Providing all three IDs (`componentId`, `projectId`, `configId`) generates a cipher starting with +`KBC::ConfigSecure::`, limiting decryption to a specific configuration. This is useful for preventing the copying of configurations. + +- Using only `projectId` yields a cipher that begins with `KBC::ProjectWideSecure::`, decryptable across the project's configurations. +This cipher type helps encrypt information shared across multiple components, e.g., SSH tunnel settings. + +- Adding `branchType` restricts the encryption to the default production branch or to development branches. This means an encrypted value with this setting cannot be moved between production and development branches or vice versa. It is not possible to encrypt a value for just one development branch. + + - Using `branchType` with `componentId` and `projectId` results in a cipher beginning with `KBC::BranchTypeSecure::`. This allows decryption either in the production or in the development configuration of the specified component in the project. + + - Using `branchType` with all three IDs (`componentId`, `projectId`, `configId`) creates a cipher that starts with `KBC::BranchTypeConfigSecure::`. It can only be decrypted within a specific production or development component configuration in a specific project. + + - Using `branchType` with `projectId` creates a cipher beginning with `KBC::ProjectWideBranchTypeSecure::`. This cipher allows decryption either in the production or in the development configurations in the project. + +The following rules apply to all ciphers: + +- Providing only a `configId` without a `projectId` is not allowed. Similarly, providing only `branchType` without `projectId` is also not allowed. +- Cipher decryption is only possible in the [region](/overview/api/#regions-and-endpoints) where the cipher was created. For example, ciphers with prefixes `KBC::ProjectSecureKV::` (Azure) or `KBC::ProjectSecureGKMS::` (GCP), instead of `KBC::ProjectSecure::` (AWS), use the same business logic but are specific to their region and technology and are not interchangeable. +- There is no decryption API; the cipher is decrypted internally before a component is run. +- Ciphering a value that is already encrypted does not change its encryption. +- There is no way to retrieve the component, project, configuration ID, or branch type from the cipher. +- The IDs referenced during cipher creation do not need to exist then. For example, you can create a cipher for a component not yet registered, which will start working as soon as the component is registered. Similarly, ciphers can be created for projects and configurations without access to them. + +By default, values encrypted in component configurations are encrypted using the `KBC::ProjectSecure::` cipher, meaning +the cipher is not transferable between regions, components, or projects. It is transferable between +different configurations of the same component within the project where it was created. If you create a configuration containing `KBC::ConfigSecure::` ciphers, +note that the configuration will not work when copied. diff --git a/src/content/docs/storage/files/index.md b/src/content/docs/storage/files/index.md index aeafc8c5a..829ee1371 100644 --- a/src/content/docs/storage/files/index.md +++ b/src/content/docs/storage/files/index.md @@ -66,7 +66,7 @@ Such a URL is valid for the entire validity of the file itself (either 15 days o In some cases, the file may be **sliced**. When you encounter a *sliced file*, you will obtain a [JSON](https://en.wikipedia.org/wiki/JSON) manifest file instead of the actual file. This can happen for some [exported or imported tables](/storage/tables/uploads/) from Storage or files which are particularly large. -Merging a sliced file requires a [substantial effort](https://developers.keboola.com/integrate/storage/api/import-export/#working-with-sliced-files). +Merging a sliced file requires a [substantial effort](/integrate/storage/api/import-export/#working-with-sliced-files). ## Limits The maximum allowed size of an uploaded file is currently 2 GB (2,048,000,000 bytes exactly). diff --git a/src/content/docs/storage/index.md b/src/content/docs/storage/index.md index 9ebd8708b..c516ddcda 100644 --- a/src/content/docs/storage/index.md +++ b/src/content/docs/storage/index.md @@ -14,7 +14,7 @@ By default all new [Pay As You Go projects](/management/payg-project/) use the B As with all other Keboola components, everything that can be done through the UI can be also done programmatically via the [Storage API](https://api.keboola.com/?service=storage). -See our [developers guide](https://developers.keboola.com/integrate/storage/) to learn more. +See our [developers guide](/integrate/storage/) to learn more. Every Storage operation must be authorized via a [token](/management/project/tokens/). It is also recorded in [Events](/management/project/tokens/#token-events) and [Jobs](/management/jobs/). diff --git a/src/content/docs/storage/tables/data-types/index.md b/src/content/docs/storage/tables/data-types/index.md index 2b475f69b..87fc25a36 100644 --- a/src/content/docs/storage/tables/data-types/index.md +++ b/src/content/docs/storage/tables/data-types/index.md @@ -64,7 +64,7 @@ To ensure typed tables are imported correctly into Storage, define your table in ## Base Types Source data types are mapped to a destination using a **base type**. The current base types are `STRING`, `INTEGER`, `NUMERIC`, `FLOAT`, `BOOLEAN`, `DATE`, and `TIMESTAMP`. For example, a MySQL extractor may store a column with the data type `BIGINT`. This type is mapped to the `INTEGER` base type, ensuring high interoperability between components. -For detailed mappings, please refer to the [conversion table](https://developers.keboola.com/extend/common-interface/manifest-files/out-tables-manifests-native-types/#data-type-conversions). You can also view the extracted data types in the [storage table](/storage/tables/) detail. +For detailed mappings, please refer to the [conversion table](/extend/common-interface/manifest-files/out-tables-manifests-native-types/#data-type-conversions). You can also view the extracted data types in the [storage table](/storage/tables/) detail. ### How to Define Data Types diff --git a/src/content/docs/storage/tables/uploads.md b/src/content/docs/storage/tables/uploads.md index 62c032d27..fe0ab349d 100644 --- a/src/content/docs/storage/tables/uploads.md +++ b/src/content/docs/storage/tables/uploads.md @@ -17,7 +17,7 @@ Every time a table is **exported** from Storage, the process is reversed: first, created in *Files* and then it is actually downloaded from there. This does not apply when exporting Storage tables manually though. Beware, however, that due to the nature of database exports, the exported table may be **sliced** and require -[substantial effort to reconstruct](https://developers.keboola.com/integrate/storage/api/import-export/#working-with-sliced-files). +[substantial effort to reconstruct](/integrate/storage/api/import-export/#working-with-sliced-files). To make sure your tables are exported as merged files, always use the **Export** feature in the **Action** tab of the table detail: diff --git a/src/content/docs/transformations/code-patterns/index.md b/src/content/docs/transformations/code-patterns/index.md index a4b6f6216..be2a7e5b5 100644 --- a/src/content/docs/transformations/code-patterns/index.md +++ b/src/content/docs/transformations/code-patterns/index.md @@ -85,7 +85,7 @@ The form is generated dynamically based on the component specification in the Ke Generated code is read only, it cannot be adjusted manually. It is (re)generated by clicking the **Regenerate Code** button. -This calls the [Generate Action](https://developers.keboola.com/extend/component/code-patterns/interface#generate-action) +This calls the [Generate Action](/extend/component/code-patterns/interface/#generate-action) on the code pattern component with the actual parameters. The result is then saved and displayed. ![Screenshot -- Generated Code](/transformations/code-patterns/overview-6-code.png) diff --git a/src/content/docs/transformations/dbt/cli/cli.md b/src/content/docs/transformations/dbt/cli/cli.md index e281c11dd..2c043bda2 100644 --- a/src/content/docs/transformations/dbt/cli/cli.md +++ b/src/content/docs/transformations/dbt/cli/cli.md @@ -8,9 +8,9 @@ Video: ## Local Development -Let's set up the local development with [Keboola CLI](https://developers.keboola.com/cli/). +Let's set up the local development with [Keboola CLI](/cli/keboola-as-code/). -It is easy on Mac with [homebrew](https://docs.brew.sh/Installation.html) support (other platforms covered in the [documentation](https://developers.keboola.com/cli/installation/)): +It is easy on Mac with [homebrew](https://docs.brew.sh/Installation.html) support (other platforms covered in the [documentation](/cli/keboola-as-code/installation/)): ```bash brew tap keboola/keboola-cli diff --git a/src/content/docs/transformations/index.md b/src/content/docs/transformations/index.md index 269db4ae5..7e095c1fd 100644 --- a/src/content/docs/transformations/index.md +++ b/src/content/docs/transformations/index.md @@ -193,7 +193,7 @@ Python and R transformations. Not available - API Interface + API Interface ✓ @@ -213,9 +213,9 @@ Python and R transformations. ### Transformations Transformations behave like any other [component](/components/). This means that they use the -standard [API](https://developers.keboola.com/integrate/storage/api/configurations/) to manipulate +standard [API](/integrate/storage/api/configurations/) to manipulate and run configurations and that creating your own -[transformation components](https://developers.keboola.com/extend/component/) is possible. +[transformation components](/extend/component/) is possible. Transformations support [sharing pieces of code](/transformations/variables/#shared-code), encouraging users to create reusable blocks of code. They also support diff --git a/src/content/docs/transformations/mappings/index.md b/src/content/docs/transformations/mappings/index.md index fc0b431c7..f17986969 100644 --- a/src/content/docs/transformations/mappings/index.md +++ b/src/content/docs/transformations/mappings/index.md @@ -43,7 +43,7 @@ While this makes some operations seemingly unnecessarily complicated, it ensures are repeatable and you can't inadvertently overwrite data in the project Storage. For ad-hoc operations, we recommend you use **[workspaces](/workspace/)**. For bulk operations, consider taking advantage of **[variables](/transformations/variables/)** -and [programmatic automation](https://developers.keboola.com/automate/). +and [programmatic automation](/automate/). ## Input Mapping Both [Storage tables](/storage/tables/) and [Storage files](/storage/files/) can be used in the input mapping of a transformation. @@ -277,7 +277,7 @@ simplify the transformation script implementation (no need to worry about cleanu Keep in mind that every table or file specified in the output mapping must be physically present in the staging area. A missing source table for the output mapping is an error. This is important when the results of a transformation are empty --- you have to ensure that an empty table or an empty file (with a header or -a [manifest](https://developers.keboola.com/extend/common-interface/manifest-files/#dataouttables-manifests)) is created. +a [manifest](/extend/common-interface/manifest-files/#dataouttables-manifests)) is created. ### Table Output Mapping Depending on the transformation backend, the table output mapping process can do the following: diff --git a/src/content/docs/transformations/python-plain/index.md b/src/content/docs/transformations/python-plain/index.md index 61e32894a..9f9190a88 100644 --- a/src/content/docs/transformations/python-plain/index.md +++ b/src/content/docs/transformations/python-plain/index.md @@ -11,10 +11,10 @@ redirect_from: other operations are too difficult. Common data operations like joining, sorting, or grouping are still easier and faster to do in [SQL Transformations](/transformations/#backends). -***Warning:** Python transformations have **no facility for encrypting secrets**. Any credential you place in transformation code — API keys, passwords, tokens, connection strings — is stored as **plaintext** in the configuration. It is not encrypted at rest, it is readable by anyone with access to the project's configuration, and it is included when the configuration is processed by features such as the AI **Generate description**. **Do not put credentials in transformation code.** Instead, store them in the [Custom Python](/components/applications/custom-python/) application, where any parameter whose key starts with `#` is [encrypted](https://developers.keboola.com/overview/encryption/) and made available to your code as an environment variable at runtime.* +***Warning:** Python transformations have **no facility for encrypting secrets**. Any credential you place in transformation code — API keys, passwords, tokens, connection strings — is stored as **plaintext** in the configuration. It is not encrypted at rest, it is readable by anyone with access to the project's configuration, and it is included when the configuration is processed by features such as the AI **Generate description**. **Do not put credentials in transformation code.** Instead, store them in the [Custom Python](/components/applications/custom-python/) application, where any parameter whose key starts with `#` is [encrypted](/overview/encryption/) and made available to your code as an environment variable at runtime.* ## Environment -The Python script is running in an isolated [environment](https://developers.keboola.com/extend/#component). +The Python script is running in an isolated [environment](/extend/#component). The Python version is updated regularly, few weeks after the official release. The update is always announced on the [status page](https://keboolastatus.com/). @@ -27,7 +27,7 @@ The Python script itself will be compiled to `/data/script.py`. To access your [mapped input and output](/transformations/mappings/) tables, use relative (`in/tables/file.csv`, `out/tables/file.csv`) or absolute (`/data/in/tables/file.csv`, `/data/out/tables/file.csv`) paths. To access downloaded files, use the `in/files/` or `/data/in/files/` path. If you want to dig really deep, -have a look at the [full Common Interface specification](https://developers.keboola.com/extend/common-interface/). +have a look at the [full Common Interface specification](/extend/common-interface/). Temporary files can be written to a `/tmp/` folder. Do not use the `/data/` folder for those files you do not wish to exchange with Keboola. ## Python Script Requirements diff --git a/src/content/docs/transformations/r-plain/index.md b/src/content/docs/transformations/r-plain/index.md index b4fd5a2b1..81323b471 100644 --- a/src/content/docs/transformations/r-plain/index.md +++ b/src/content/docs/transformations/r-plain/index.md @@ -14,7 +14,7 @@ other operations are too difficult. Common data operations like joining, sorting faster to do in [SQL Transformations](/transformations/#backends). ## Environment -The R script is executed in an isolated [environment](https://developers.keboola.com/extend/#component). +The R script is executed in an isolated [environment](/extend/#component). The current R version is **4.0.5**, however it is possible to switch your configuration to run on other versions if available. ![Screenshot - Change Backend](/transformations/r-plain/change-backend.png) @@ -32,7 +32,7 @@ The R script itself will be compiled to `/data/script.R`. To access your [mapped input and output](/transformations/mappings/) tables, use relative (`in/tables/file.csv`, `out/tables/file.csv`) or absolute (`/data/in/tables/file.csv`, `/data/out/tables/file.csv`) paths. To access downloaded files, use the `in/files/` or `/data/in/files/` path. If you want to dig really deep, -have a look at the [full Common Interface specification](https://developers.keboola.com/extend/common-interface/). +have a look at the [full Common Interface specification](/extend/common-interface/). Temporary files can be written to a `/tmp/` folder. Do not use the `/data/` folder for those files you do not wish to exchange with Keboola. ## R Script Requirements diff --git a/src/content/docs/transformations/snowflake-plain/index.md b/src/content/docs/transformations/snowflake-plain/index.md index 90733d0df..6281fb5a0 100644 --- a/src/content/docs/transformations/snowflake-plain/index.md +++ b/src/content/docs/transformations/snowflake-plain/index.md @@ -219,7 +219,7 @@ CREATE TABLE "out" AS Do not use `ALTER SESSION` queries to modify the default timestamp format, as the loading and unloading sessions are separate from your transformation/sandbox session and the format may change unexpectedly. -**Important:** In the AWS US Keboola [region](https://developers.keboola.com/overview/api/#regions-and-endpoints) +**Important:** In the AWS US Keboola [region](/overview/api/#regions-and-endpoints) (connection.keboola.com), the following [Snowflake default](https://docs.snowflake.com/en/sql-reference/parameters) parameters are overridden: diff --git a/src/content/docs/tutorial/index.md b/src/content/docs/tutorial/index.md index 3da71ab35..670394916 100644 --- a/src/content/docs/tutorial/index.md +++ b/src/content/docs/tutorial/index.md @@ -51,7 +51,7 @@ For a deeper exploration of Keboola features, aligning with real-world usage, co - Explore how to perform ad-hoc data analysis, allowing flexibility in interacting with arbitrary data. 5. [**Development branches**](/tutorial/branches/) - Learn how to safely modify a running project using development branches. -6. [**Command-line interface (CLI)**](https://developers.keboola.com/cli/) +6. [**Command-line interface (CLI)**](/cli/keboola-as-code/) - Operate a project efficiently using the Keboola command-line tool. These advanced steps will provide you with a comprehensive understanding of Keboola's capabilities and their practical application in real-world scenarios. diff --git a/src/content/docs/tutorial/onboarding/index.md b/src/content/docs/tutorial/onboarding/index.md index f986eb242..4d50a0f1a 100644 --- a/src/content/docs/tutorial/onboarding/index.md +++ b/src/content/docs/tutorial/onboarding/index.md @@ -52,7 +52,7 @@ It's time to get practical: - Start with our [Keboola Introduction](https://academy.keboola.com/courses/introduction-2023). - Check out [General Best Practices](https://academy.keboola.com/courses/best-practices-2023). - Solve problems with our [Debugging Techniques](https://academy.keboola.com/courses/debug-techniques). -- Developers, see this [video](https://www.youtube.com/watch?v=IhET2hDD_1w) and [documentation](https://developers.keboola.com/extend/) on making new components. Learn more in our [academy lessons](https://academy.keboola.com/courses/common-components-and-processors). +- Developers, see this [video](https://www.youtube.com/watch?v=IhET2hDD_1w) and [documentation](/extend/) on making new components. Learn more in our [academy lessons](https://academy.keboola.com/courses/common-components-and-processors). ## Cheat Sheet: Embracing Best Practices Mastering Keboola means knowing how to set up components, automate workflows, and more. diff --git a/src/sidebar.mjs b/src/sidebar.mjs index c4caa8432..7c26a1b7e 100644 --- a/src/sidebar.mjs +++ b/src/sidebar.mjs @@ -513,6 +513,414 @@ export const sidebar = [ { slug: "ai/mcp-server" }, ], }, + { + label: "Developer Docs", + collapsed: true, + items: [ + { label: "Overview", slug: "overview/api" }, + { slug: "overview/api" }, + { slug: "overview/encryption" }, + { + label: "Extending Keboola", + collapsed: true, + items: [ + { label: "Overview", slug: "extend" }, + { + label: "Components", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/component" }, + { + label: "Tutorial", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/component/tutorial" }, + { slug: "extend/component/tutorial/input-mapping" }, + { slug: "extend/component/tutorial/output-mapping" }, + { slug: "extend/component/tutorial/configuration" }, + { slug: "extend/component/tutorial/processors" }, + { slug: "extend/component/tutorial/debugging" }, + ], + }, + { slug: "extend/component/processors" }, + { + label: "Code Patterns", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/component/code-patterns" }, + { slug: "extend/component/code-patterns/interface" }, + { slug: "extend/component/code-patterns/tutorial" }, + ], + }, + { + label: "Implementation Notes", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/component/implementation" }, + { slug: "extend/component/implementation/php" }, + { slug: "extend/component/implementation/python" }, + { slug: "extend/component/implementation/r" }, + ], + }, + { slug: "extend/component/running" }, + { + label: "UI Options", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/component/ui-options" }, + { + label: "Configuration Schema", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/component/ui-options/configuration-schema" }, + { slug: "extend/component/ui-options/configuration-schema/examples" }, + { slug: "extend/component/ui-options/configuration-schema/sync-action-examples" }, + ], + }, + { slug: "extend/component/ui-options/default-configuration" }, + ], + }, + { slug: "extend/component/deployment" }, + ], + }, + { + label: "Generic Extractor", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-extractor" }, + { + label: "Generic Extractor Tutorial", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-extractor/tutorial" }, + { slug: "extend/generic-extractor/tutorial/rest" }, + { slug: "extend/generic-extractor/tutorial/json" }, + { slug: "extend/generic-extractor/tutorial/basic" }, + { slug: "extend/generic-extractor/tutorial/pagination" }, + { slug: "extend/generic-extractor/tutorial/jobs" }, + { slug: "extend/generic-extractor/tutorial/mapping" }, + ], + }, + { + label: "Configuration", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-extractor/configuration" }, + { + label: "API Configuration", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-extractor/configuration/api" }, + { + label: "Pagination", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-extractor/configuration/api/pagination" }, + { slug: "extend/generic-extractor/configuration/api/pagination/response-url" }, + { slug: "extend/generic-extractor/configuration/api/pagination/response-param" }, + { slug: "extend/generic-extractor/configuration/api/pagination/offset" }, + { slug: "extend/generic-extractor/configuration/api/pagination/pagenum" }, + { slug: "extend/generic-extractor/configuration/api/pagination/cursor" }, + { slug: "extend/generic-extractor/configuration/api/pagination/multiple" }, + ], + }, + { + label: "Authentication", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-extractor/configuration/api/authentication" }, + { slug: "extend/generic-extractor/configuration/api/authentication/query" }, + { slug: "extend/generic-extractor/configuration/api/authentication/basic" }, + { slug: "extend/generic-extractor/configuration/api/authentication/bearer_token" }, + { slug: "extend/generic-extractor/configuration/api/authentication/api_key" }, + { slug: "extend/generic-extractor/configuration/api/authentication/login" }, + { slug: "extend/generic-extractor/configuration/api/authentication/oauth_cc" }, + { slug: "extend/generic-extractor/configuration/api/authentication/oauth10" }, + { slug: "extend/generic-extractor/configuration/api/authentication/oauth20" }, + { slug: "extend/generic-extractor/configuration/api/authentication/oauth20-login" }, + ], + }, + ], + }, + { + label: "Extraction Configuration", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-extractor/configuration/config" }, + { + label: "Jobs", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-extractor/configuration/config/jobs" }, + { slug: "extend/generic-extractor/configuration/config/jobs/children" }, + ], + }, + { slug: "extend/generic-extractor/configuration/config/mappings" }, + ], + }, + { slug: "extend/generic-extractor/configuration/iterations" }, + { slug: "extend/generic-extractor/configuration/ssh-proxy" }, + ], + }, + { slug: "extend/generic-extractor/map" }, + { slug: "extend/generic-extractor/functions" }, + { slug: "extend/generic-extractor/incremental" }, + { slug: "extend/generic-extractor/running" }, + { slug: "extend/generic-extractor/publish" }, + ], + }, + { + label: "Generic Writer", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/generic-writer" }, + { slug: "extend/generic-writer/configuration" }, + { slug: "extend/generic-writer/configuration-examples" }, + ], + }, + { + label: "Common Interface", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/common-interface" }, + { slug: "extend/common-interface/folders" }, + { slug: "extend/common-interface/config-file" }, + { slug: "extend/common-interface/environment" }, + { + label: "Manifest Files", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/common-interface/manifest-files" }, + { slug: "extend/common-interface/manifest-files/in-tables-manifests" }, + { slug: "extend/common-interface/manifest-files/in-files-manifests" }, + { slug: "extend/common-interface/manifest-files/in-files-s3-staging" }, + { slug: "extend/common-interface/manifest-files/in-files-abs-staging" }, + { slug: "extend/common-interface/manifest-files/out-tables-manifests" }, + { slug: "extend/common-interface/manifest-files/out-tables-manifests-native-types" }, + { slug: "extend/common-interface/manifest-files/out-files-manifests" }, + ], + }, + { slug: "extend/common-interface/oauth" }, + { slug: "extend/common-interface/actions" }, + { slug: "extend/common-interface/logging" }, + { slug: "extend/common-interface/development-branches" }, + ], + }, + { slug: "extend/job-queue" }, + { + label: "Publishing Component", + collapsed: true, + items: [ + { label: "Overview", slug: "extend/publish" }, + { slug: "extend/publish/checklist" }, + ], + }, + ], + }, + { + label: "Integration", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate" }, + { + label: "Storage", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate/storage" }, + { slug: "integrate/storage/php-client" }, + { slug: "integrate/storage/r-client" }, + { slug: "integrate/storage/python-client" }, + { slug: "integrate/storage/docker-cli-client" }, + { + label: "Using API", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate/storage/api" }, + { slug: "integrate/storage/api/configurations" }, + { slug: "integrate/storage/api/importer" }, + { slug: "integrate/storage/api/import-export" }, + { slug: "integrate/storage/api/tde-exporter" }, + ], + }, + ], + }, + { slug: "integrate/jobs" }, + { + label: "Variables", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate/variables" }, + { slug: "integrate/variables/tutorial" }, + ], + }, + { + label: "Artifacts", + collapsed: true, + items: [ + { label: "Overview", slug: "integrate/artifacts" }, + { slug: "integrate/artifacts/tutorial" }, + ], + }, + ], + }, + { + label: "Automation/Common Tasks", + collapsed: true, + items: [ + { label: "Overview", slug: "automate" }, + { slug: "automate/run-job" }, + { slug: "automate/run-orchestration" }, + { slug: "automate/set-schedule" }, + ], + }, + { + label: "Keboola as Code CLI", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code" }, + { slug: "cli/keboola-as-code/installation" }, + { slug: "cli/keboola-as-code/getting-started" }, + { slug: "cli/keboola-as-code/structure" }, + { + label: "Commands", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands" }, + { slug: "cli/keboola-as-code/commands/help" }, + { slug: "cli/keboola-as-code/commands/status" }, + { + label: "sync", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/sync" }, + { slug: "cli/keboola-as-code/commands/sync/init" }, + { slug: "cli/keboola-as-code/commands/sync/pull" }, + { slug: "cli/keboola-as-code/commands/sync/push" }, + { slug: "cli/keboola-as-code/commands/sync/diff" }, + ], + }, + { + label: "ci", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/ci" }, + { slug: "cli/keboola-as-code/commands/ci/workflows" }, + ], + }, + { + label: "local", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/local" }, + { + label: "create", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/local/create" }, + { slug: "cli/keboola-as-code/commands/local/create/config" }, + { slug: "cli/keboola-as-code/commands/local/create/row" }, + ], + }, + { slug: "cli/keboola-as-code/commands/local/persist" }, + { slug: "cli/keboola-as-code/commands/local/encrypt" }, + { + label: "validate", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/local/validate" }, + { slug: "cli/keboola-as-code/commands/local/validate/config" }, + { slug: "cli/keboola-as-code/commands/local/validate/row" }, + { slug: "cli/keboola-as-code/commands/local/validate/schema" }, + ], + }, + { slug: "cli/keboola-as-code/commands/local/fix-paths" }, + ], + }, + { + label: "remote", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/remote" }, + { + label: "create", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/remote/create" }, + { slug: "cli/keboola-as-code/commands/remote/create/branch" }, + { slug: "cli/keboola-as-code/commands/remote/create/bucket" }, + ], + }, + { + label: "file", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/remote/file" }, + { slug: "cli/keboola-as-code/commands/remote/file/download" }, + { slug: "cli/keboola-as-code/commands/remote/file/upload" }, + ], + }, + { + label: "job", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/remote/job" }, + { slug: "cli/keboola-as-code/commands/remote/job/run" }, + ], + }, + { + label: "table", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/remote/table" }, + { slug: "cli/keboola-as-code/commands/remote/table/create" }, + { slug: "cli/keboola-as-code/commands/remote/table/upload" }, + { slug: "cli/keboola-as-code/commands/remote/table/download" }, + { slug: "cli/keboola-as-code/commands/remote/table/preview" }, + { slug: "cli/keboola-as-code/commands/remote/table/detail" }, + { slug: "cli/keboola-as-code/commands/remote/table/import" }, + { slug: "cli/keboola-as-code/commands/remote/table/unload" }, + ], + }, + { + label: "workspace", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/remote/workspace" }, + { slug: "cli/keboola-as-code/commands/remote/workspace/create" }, + { slug: "cli/keboola-as-code/commands/remote/workspace/delete" }, + { slug: "cli/keboola-as-code/commands/remote/workspace/detail" }, + { slug: "cli/keboola-as-code/commands/remote/workspace/list" }, + ], + }, + ], + }, + { + label: "dbt", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/dbt" }, + { slug: "cli/keboola-as-code/commands/dbt/init" }, + { + label: "generate", + collapsed: true, + items: [ + { label: "Overview", slug: "cli/keboola-as-code/commands/dbt/generate" }, + { slug: "cli/keboola-as-code/commands/dbt/generate/profile" }, + { slug: "cli/keboola-as-code/commands/dbt/generate/sources" }, + { slug: "cli/keboola-as-code/commands/dbt/generate/env" }, + ], + }, + ], + }, + ], + }, + { slug: "cli/keboola-as-code/github-integration" }, + { slug: "cli/keboola-as-code/devops-use-cases" }, + { slug: "cli/keboola-as-code/dbt" }, + ], + }, + ], + }, { label: "External Integrations", collapsed: true, diff --git a/src/styles/custom.css b/src/styles/custom.css index b418818e4..ad40f4c01 100644 --- a/src/styles/custom.css +++ b/src/styles/custom.css @@ -1802,3 +1802,55 @@ nav.sidebar { #ask-kai-drawer .ak-shell { width: 100vw; } #ak-fab { bottom: 20px; right: 16px; } } + +/* ============================================================================ + TRANSITIONAL (phase 1 of the dev-docs migration) — REMOVE IN PHASE 2. + Marks sidebar entries migrated from developers.keboola.com with an accent + dot + a legend at the bottom of the sidebar. Scoped purely by URL prefix; + dies together with the temporary "Developer Docs" group. + ========================================================================== */ + +/* dot on migrated pages */ +#starlight__sidebar a[href^="/extend/"]::before, +#starlight__sidebar a[href^="/integrate/"]::before, +#starlight__sidebar a[href^="/automate/"]::before, +#starlight__sidebar a[href^="/cli/keboola-as-code"]::before, +#starlight__sidebar a[href^="/overview/api/"]::before, +#starlight__sidebar a[href^="/overview/encryption/"]::before { + content: ""; + display: inline-block; + flex: none; + width: 6px; + height: 6px; + border-radius: 50%; + background: var(--sl-color-accent); + margin-right: 0.45rem; + align-self: center; +} + +/* dot on groups whose subtree is migrated (incl. the Developer Docs root) */ +#starlight__sidebar details:has(a[href^="/extend/"]) > summary .group-label::before, +#starlight__sidebar details:has(a[href^="/integrate/"]) > summary .group-label::before, +#starlight__sidebar details:has(a[href^="/automate/"]) > summary .group-label::before, +#starlight__sidebar details:has(a[href^="/cli/keboola-as-code"]) > summary .group-label::before { + content: ""; + display: inline-block; + flex: none; + width: 6px; + height: 6px; + border-radius: 50%; + background: var(--sl-color-accent); + margin-right: 0.45rem; + align-self: center; +} + +/* legend at the bottom of the sidebar */ +#starlight__sidebar .sidebar-content::after { + content: "\2022 migrated from developers.keboola.com"; + display: block; + margin-top: auto; + padding: 0.75rem 0.25rem 0.25rem; + border-top: 1px solid var(--sl-color-hairline-light, var(--sl-color-gray-5)); + font-size: var(--sl-text-xs); + color: var(--sl-color-gray-3); +}