Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions plugins/kbagent/skills/kbagent/references/gotchas.md
Original file line number Diff line number Diff line change
Expand Up @@ -2090,6 +2090,17 @@ config, the retry fires, and the retry destroys it for good.
do NOT appear in the Apps UI. The command now keeps only
`componentId == keboola.data-apps` (items missing `componentId` are kept
defensively); the JSON envelope carries `component_id` per app.
- **`data-app list` pages through the whole `GET /apps` collection
*(since vNEXT, #798)*.** The endpoint is paginated (default page = 100
items) and the workspace/data-app mix is filtered CLIENT-side. Before
vNEXT kbagent read only that first page, so a project with many
workspaces could report "No data apps found." (or a partial list) while
holding dozens of data apps further down the collection -- one reporter's
first data app was item #258 of 1,114. The same short read also made
`sync pull` miss the runtime type of those apps. kbagent now pages with
`limit`/`offset` until a short page. On an older version, an empty or
short `data-app list` is NOT evidence that the project has no data apps:
cross-check with `config list --component-id keboola.data-apps`.
- **`secrets-remove` is idempotent.** Removing a key that isn't set is
exit 0 with `removed: 0`, `not_found: [<derived env-var name>]`. The
Storage version is not bumped on a no-op. Do NOT script around this
Expand Down
8 changes: 8 additions & 0 deletions src/keboola_agent_cli/constants.py
Original file line number Diff line number Diff line change
Expand Up @@ -710,6 +710,14 @@ def _resolve_app_name() -> str:
QUERY_RESULTS_DEFAULT_LIMIT: int = 500 # default --limit for `workspace query` fast path
QUERY_RESULTS_PAGE_SIZE: int = 500 # rows per /results page (API requires 100..100000)

# --- Data Science API (data apps) ---
# GET /apps is paginated (default page = 100 items) and mixes workspace
# deployments with data apps, so list_apps() pages with limit/offset until a
# short page (#798). MAX_PAGES only guards against a server that ignores
# ``offset`` (500 * 200 = 100k deployments, far beyond any real project).
DATA_SCIENCE_APPS_PAGE_SIZE: int = 500 # items per GET /apps page
DATA_SCIENCE_APPS_MAX_PAGES: int = 200 # safety cap on pages fetched by list_apps()

# --- Workspace Defaults ---
DEFAULT_WORKSPACE_BACKEND: str = "snowflake"

Expand Down
47 changes: 39 additions & 8 deletions src/keboola_agent_cli/data_science_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,8 @@
validation):

POST /apps -> 201, {id, configId, ...}
GET /apps -> 200, [{id, configId, state, desiredState, url}, ...]
GET /apps?limit=N&offset=M -> 200, [{id, configId, state, desiredState, url}, ...]
(paginated; default page = 100)
GET /apps/{id} -> 200, full deployment record
PATCH /apps/{id} -> 200, deployment record (only
desiredState / configVersion /
Expand All @@ -43,7 +44,11 @@

import httpx

from .constants import DEFAULT_TIMEOUT
from .constants import (
DATA_SCIENCE_APPS_MAX_PAGES,
DATA_SCIENCE_APPS_PAGE_SIZE,
DEFAULT_TIMEOUT,
)
from .http_base import BaseHttpClient

logger = logging.getLogger(__name__)
Expand Down Expand Up @@ -76,16 +81,42 @@ def __exit__(self, *args: object) -> None:
self.close()

def list_apps(self) -> list[dict[str, Any]]:
"""Return the thin index of data apps in the project (no body filter).
"""Return the thin index of ALL deployments in the project (no body filter).

The Data Science API scopes responses by the token's project; there
is no ``branchId`` query parameter on the list endpoint.

``GET /apps`` is paginated: without ``limit``/``offset`` it returns
only a default first page (100 items) that mixes workspace
deployments (``keboola.sandboxes``) with data apps
(``keboola.data-apps``). Callers filter client-side, so a project with
many workspaces could have every data app beyond that first page and
``data-app list`` reported "No data apps found." (#798). We therefore
page with ``limit``/``offset`` until a short (or empty) page.
"""
response = self._do_request("GET", "/apps")
body = response.json()
# Some stacks wrap the list in {"data": [...]}; fall back gracefully.
apps = (body.get("data") or body.get("apps") or []) if isinstance(body, dict) else body
return apps if isinstance(apps, list) else []
apps: list[dict[str, Any]] = []
page_size = DATA_SCIENCE_APPS_PAGE_SIZE
for page in range(DATA_SCIENCE_APPS_MAX_PAGES):
response = self._do_request(
"GET", "/apps", params={"limit": page_size, "offset": page * page_size}
)
body = response.json()
# Some stacks wrap the list in {"data": [...]}; fall back gracefully.
items = (body.get("data") or body.get("apps") or []) if isinstance(body, dict) else body
if not isinstance(items, list):
break
apps.extend(items)
if len(items) < page_size:
break
else:
# Guard against a server that ignores ``offset`` and keeps
# returning full pages -- never loop forever, but say so.
logger.warning(
"GET /apps: stopped after %d pages of %d; the listing may be incomplete",
DATA_SCIENCE_APPS_MAX_PAGES,
page_size,
)
return apps

def get_app(self, app_id: str) -> dict[str, Any]:
"""Fetch a single deployment record by numeric app id."""
Expand Down
127 changes: 127 additions & 0 deletions tests/test_data_app_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -1904,3 +1904,130 @@ def test_tail_app_logs_no_params_sends_clean_url(self, httpx_mock) -> None:
text = client.tail_app_logs("42")

assert text == "full buffer\n"


# ---------------------------------------------------------------------------
# data-app list pagination (client HTTP layer via httpx_mock) -- issue #798
# ---------------------------------------------------------------------------


class TestListAppsPaginationClient:
"""``GET /apps`` is paginated and mixes workspaces with data apps.

Before #798 ``list_apps`` fetched only the server's default first page, so
a project whose first deployments were all workspaces (``keboola.sandboxes``)
reported "No data apps found." even with 139 data apps further down.
"""

DATA_SCIENCE_BASE = "https://data-science.keboola.com"

@staticmethod
def _client():
from keboola_agent_cli.data_science_client import DataScienceClient

return DataScienceClient(stack_url="https://connection.keboola.com", token="901-test-token")

@staticmethod
def _sandboxes(n: int, start: int = 0) -> list[dict[str, Any]]:
return [
{"id": str(start + i), "componentId": "keboola.sandboxes", "type": "snowflake"}
for i in range(n)
]

def test_single_short_page_makes_one_request(self, httpx_mock) -> None:
from keboola_agent_cli.constants import DATA_SCIENCE_APPS_PAGE_SIZE as size

httpx_mock.add_response(
url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset=0",
json=[{"id": "1", "componentId": "keboola.data-apps"}],
)
with self._client() as client:
apps = client.list_apps()

assert [a["id"] for a in apps] == ["1"]
assert len(httpx_mock.get_requests()) == 1

def test_pages_until_short_page_and_returns_later_data_apps(self, httpx_mock) -> None:
from keboola_agent_cli.constants import DATA_SCIENCE_APPS_PAGE_SIZE as size

data_apps = [
{"id": "d1", "componentId": "keboola.data-apps", "configId": "c1"},
{"id": "d2", "componentId": "keboola.data-apps", "configId": "c2"},
]
httpx_mock.add_response(
url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset=0",
json=self._sandboxes(size),
)
httpx_mock.add_response(
url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset={size}",
json=[*self._sandboxes(3, start=size), *data_apps],
)
with self._client() as client:
apps = client.list_apps()

assert len(apps) == size + 5
assert [a["id"] for a in apps if a["componentId"] == "keboola.data-apps"] == ["d1", "d2"]
assert [dict(r.url.params) for r in httpx_mock.get_requests()] == [
{"limit": str(size), "offset": "0"},
{"limit": str(size), "offset": str(size)},
]

def test_exact_full_last_page_is_followed_by_empty_page(self, httpx_mock) -> None:
from keboola_agent_cli.constants import DATA_SCIENCE_APPS_PAGE_SIZE as size

httpx_mock.add_response(
url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset=0",
json=self._sandboxes(size),
)
httpx_mock.add_response(
url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset={size}",
json=[],
)
with self._client() as client:
apps = client.list_apps()

assert len(apps) == size
assert len(httpx_mock.get_requests()) == 2

def test_wrapped_data_shape_is_still_supported(self, httpx_mock) -> None:
from keboola_agent_cli.constants import DATA_SCIENCE_APPS_PAGE_SIZE as size

httpx_mock.add_response(
url=f"{self.DATA_SCIENCE_BASE}/apps?limit={size}&offset=0",
json={"data": [{"id": "1", "componentId": "keboola.data-apps"}]},
)
with self._client() as client:
assert [a["id"] for a in client.list_apps()] == ["1"]

def test_page_cap_stops_a_server_that_ignores_offset(self, httpx_mock) -> None:
httpx_mock.add_response(json=self._sandboxes(2), is_reusable=True)
with (
patch("keboola_agent_cli.data_science_client.DATA_SCIENCE_APPS_PAGE_SIZE", 2),
patch("keboola_agent_cli.data_science_client.DATA_SCIENCE_APPS_MAX_PAGES", 3),
self._client() as client,
):
apps = client.list_apps()

assert len(apps) == 6
assert len(httpx_mock.get_requests()) == 3


class TestDataAppListFiltersWorkspaces:
"""``list_data_apps`` keeps only ``keboola.data-apps`` rows from the full listing."""

def test_sandboxes_dropped_data_apps_kept(self, tmp_path: Path) -> None:
store = _make_store(tmp_path)
service, ds_mock, storage_mock, _enc = _make_service(store)
ds_mock.list_apps.return_value = [
*[
{"id": str(i), "componentId": "keboola.sandboxes", "type": "snowflake"}
for i in range(300)
],
{"id": "d1", "componentId": "keboola.data-apps", "configId": "c1", "type": "python-js"},
]
storage_mock.list_component_configs.return_value = [{"id": "c1", "name": "App"}]

result = service.list_data_apps(aliases=["prod"])

assert result["errors"] == []
assert [a["app_id"] for a in result["apps"]] == ["d1"]
Loading