diff --git a/plugins/kbagent/skills/kbagent/references/gotchas.md b/plugins/kbagent/skills/kbagent/references/gotchas.md index ed6e74fa..7582986a 100644 --- a/plugins/kbagent/skills/kbagent/references/gotchas.md +++ b/plugins/kbagent/skills/kbagent/references/gotchas.md @@ -2090,6 +2090,17 @@ config, the retry fires, and the retry destroys it for good. do NOT appear in the Apps UI. The command now keeps only `componentId == keboola.data-apps` (items missing `componentId` are kept defensively); the JSON envelope carries `component_id` per app. +- **`data-app list` pages through the whole `GET /apps` collection + *(since vNEXT, #798)*.** The endpoint is paginated (default page = 100 + items) and the workspace/data-app mix is filtered CLIENT-side. Before + vNEXT kbagent read only that first page, so a project with many + workspaces could report "No data apps found." (or a partial list) while + holding dozens of data apps further down the collection -- one reporter's + first data app was item #258 of 1,114. The same short read also made + `sync pull` miss the runtime type of those apps. kbagent now pages with + `limit`/`offset` until a short page. On an older version, an empty or + short `data-app list` is NOT evidence that the project has no data apps: + cross-check with `config list --component-id keboola.data-apps`. - **`secrets-remove` is idempotent.** Removing a key that isn't set is exit 0 with `removed: 0`, `not_found: []`. The Storage version is not bumped on a no-op. Do NOT script around this diff --git a/src/keboola_agent_cli/constants.py b/src/keboola_agent_cli/constants.py index 750ec3c9..dd67e4d0 100644 --- a/src/keboola_agent_cli/constants.py +++ b/src/keboola_agent_cli/constants.py @@ -710,6 +710,14 @@ def _resolve_app_name() -> str: QUERY_RESULTS_DEFAULT_LIMIT: int = 500 # default --limit for `workspace query` fast path QUERY_RESULTS_PAGE_SIZE: int = 500 # rows per /results page (API requires 100..100000) +# --- Data Science API (data apps) --- +# GET /apps is paginated (default page = 100 items) and mixes workspace +# deployments with data apps, so list_apps() pages with limit/offset until a +# short page (#798). MAX_PAGES only guards against a server that ignores +# ``offset`` (500 * 200 = 100k deployments, far beyond any real project). +DATA_SCIENCE_APPS_PAGE_SIZE: int = 500 # items per GET /apps page +DATA_SCIENCE_APPS_MAX_PAGES: int = 200 # safety cap on pages fetched by list_apps() + # --- Workspace Defaults --- DEFAULT_WORKSPACE_BACKEND: str = "snowflake" diff --git a/src/keboola_agent_cli/data_science_client.py b/src/keboola_agent_cli/data_science_client.py index 0cc5dfe3..50a1e33a 100644 --- a/src/keboola_agent_cli/data_science_client.py +++ b/src/keboola_agent_cli/data_science_client.py @@ -17,7 +17,8 @@ validation): POST /apps -> 201, {id, configId, ...} - GET /apps -> 200, [{id, configId, state, desiredState, url}, ...] + GET /apps?limit=N&offset=M -> 200, [{id, configId, state, desiredState, url}, ...] + (paginated; default page = 100) GET /apps/{id} -> 200, full deployment record PATCH /apps/{id} -> 200, deployment record (only desiredState / configVersion / @@ -43,7 +44,11 @@ import httpx -from .constants import DEFAULT_TIMEOUT +from .constants import ( + DATA_SCIENCE_APPS_MAX_PAGES, + DATA_SCIENCE_APPS_PAGE_SIZE, + DEFAULT_TIMEOUT, +) from .http_base import BaseHttpClient logger = logging.getLogger(__name__) @@ -76,16 +81,42 @@ def __exit__(self, *args: object) -> None: self.close() def list_apps(self) -> list[dict[str, Any]]: - """Return the thin index of data apps in the project (no body filter). + """Return the thin index of ALL deployments in the project (no body filter). The Data Science API scopes responses by the token's project; there is no ``branchId`` query parameter on the list endpoint. + + ``GET /apps`` is paginated: without ``limit``/``offset`` it returns + only a default first page (100 items) that mixes workspace + deployments (``keboola.sandboxes``) with data apps + (``keboola.data-apps``). Callers filter client-side, so a project with + many workspaces could have every data app beyond that first page and + ``data-app list`` reported "No data apps found." (#798). We therefore + page with ``limit``/``offset`` until a short (or empty) page. """ - response = self._do_request("GET", "/apps") - body = response.json() - # Some stacks wrap the list in {"data": [...]}; fall back gracefully. - apps = (body.get("data") or body.get("apps") or []) if isinstance(body, dict) else body - return apps if isinstance(apps, list) else [] + apps: list[dict[str, Any]] = [] + page_size = DATA_SCIENCE_APPS_PAGE_SIZE + for page in range(DATA_SCIENCE_APPS_MAX_PAGES): + response = self._do_request( + "GET", "/apps", params={"limit": page_size, "offset": page * page_size} + ) + body = response.json() + # Some stacks wrap the list in {"data": [...]}; fall back gracefully. + items = (body.get("data") or body.get("apps") or []) if isinstance(body, dict) else body + if not isinstance(items, list): + break + apps.extend(items) + if len(items) < page_size: + break + else: + # Guard against a server that ignores ``offset`` and keeps + # returning full pages -- never loop forever, but say so. + logger.warning( + "GET /apps: stopped after %d pages of %d; the listing may be incomplete", + DATA_SCIENCE_APPS_MAX_PAGES, + page_size, + ) + return apps def get_app(self, app_id: str) -> dict[str, Any]: """Fetch a single deployment record by numeric app id.""" diff --git a/tests/test_data_app_service.py b/tests/test_data_app_service.py index bf9d111a..e50e3f06 100644 --- a/tests/test_data_app_service.py +++ b/tests/test_data_app_service.py @@ -1904,3 +1904,130 @@ def test_tail_app_logs_no_params_sends_clean_url(self, httpx_mock) -> None: text = client.tail_app_logs("42") assert text == "full buffer\n" + + +# --------------------------------------------------------------------------- +# data-app list pagination (client HTTP layer via httpx_mock) -- issue #798 +# --------------------------------------------------------------------------- + + +class TestListAppsPaginationClient: + """``GET /apps`` is paginated and mixes workspaces with data apps. + + Before #798 ``list_apps`` fetched only the server's default first page, so + a project whose first deployments were all workspaces (``keboola.sandboxes``) + reported "No data apps found." even with 139 data apps further down. + """ + + DATA_SCIENCE_BASE = "https://data-science.keboola.com" + + @staticmethod + def _client(): + from keboola_agent_cli.data_science_client import DataScienceClient + + return DataScienceClient(stack_url="https://connection.keboola.com", token="901-test-token") + + @staticmethod + def _sandboxes(n: int, start: int = 0) -> list[dict[str, Any]]: + return [ + {"id": str(start + i), "componentId": "keboola.sandboxes", "type": "snowflake"} + for i in range(n) + ] + + def test_single_short_page_makes_one_request(self, httpx_mock) -> None: + from keboola_agent_cli.constants import DATA_SCIENCE_APPS_PAGE_SIZE as size + + httpx_mock.add_response( + url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset=0", + json=[{"id": "1", "componentId": "keboola.data-apps"}], + ) + with self._client() as client: + apps = client.list_apps() + + assert [a["id"] for a in apps] == ["1"] + assert len(httpx_mock.get_requests()) == 1 + + def test_pages_until_short_page_and_returns_later_data_apps(self, httpx_mock) -> None: + from keboola_agent_cli.constants import DATA_SCIENCE_APPS_PAGE_SIZE as size + + data_apps = [ + {"id": "d1", "componentId": "keboola.data-apps", "configId": "c1"}, + {"id": "d2", "componentId": "keboola.data-apps", "configId": "c2"}, + ] + httpx_mock.add_response( + url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset=0", + json=self._sandboxes(size), + ) + httpx_mock.add_response( + url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset={size}", + json=[*self._sandboxes(3, start=size), *data_apps], + ) + with self._client() as client: + apps = client.list_apps() + + assert len(apps) == size + 5 + assert [a["id"] for a in apps if a["componentId"] == "keboola.data-apps"] == ["d1", "d2"] + assert [dict(r.url.params) for r in httpx_mock.get_requests()] == [ + {"limit": str(size), "offset": "0"}, + {"limit": str(size), "offset": str(size)}, + ] + + def test_exact_full_last_page_is_followed_by_empty_page(self, httpx_mock) -> None: + from keboola_agent_cli.constants import DATA_SCIENCE_APPS_PAGE_SIZE as size + + httpx_mock.add_response( + url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset=0", + json=self._sandboxes(size), + ) + httpx_mock.add_response( + url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset={size}", + json=[], + ) + with self._client() as client: + apps = client.list_apps() + + assert len(apps) == size + assert len(httpx_mock.get_requests()) == 2 + + def test_wrapped_data_shape_is_still_supported(self, httpx_mock) -> None: + from keboola_agent_cli.constants import DATA_SCIENCE_APPS_PAGE_SIZE as size + + httpx_mock.add_response( + url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset=0", + json={"data": [{"id": "1", "componentId": "keboola.data-apps"}]}, + ) + with self._client() as client: + assert [a["id"] for a in client.list_apps()] == ["1"] + + def test_page_cap_stops_a_server_that_ignores_offset(self, httpx_mock) -> None: + httpx_mock.add_response(json=self._sandboxes(2), is_reusable=True) + with ( + patch("keboola_agent_cli.data_science_client.DATA_SCIENCE_APPS_PAGE_SIZE", 2), + patch("keboola_agent_cli.data_science_client.DATA_SCIENCE_APPS_MAX_PAGES", 3), + self._client() as client, + ): + apps = client.list_apps() + + assert len(apps) == 6 + assert len(httpx_mock.get_requests()) == 3 + + +class TestDataAppListFiltersWorkspaces: + """``list_data_apps`` keeps only ``keboola.data-apps`` rows from the full listing.""" + + def test_sandboxes_dropped_data_apps_kept(self, tmp_path: Path) -> None: + store = _make_store(tmp_path) + service, ds_mock, storage_mock, _enc = _make_service(store) + ds_mock.list_apps.return_value = [ + *[ + {"id": str(i), "componentId": "keboola.sandboxes", "type": "snowflake"} + for i in range(300) + ], + {"id": "d1", "componentId": "keboola.data-apps", "configId": "c1", "type": "python-js"}, + ] + storage_mock.list_component_configs.return_value = [{"id": "c1", "name": "App"}] + + result = service.list_data_apps(aliases=["prod"]) + + assert result["errors"] == [] + assert [a["app_id"] for a in result["apps"]] == ["d1"]