Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/fair-vans-visit.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"@upstash/box": patch
---

Support Jev browser actions with `model: "jev"` (also `typesafe-ai/jev` and `vercel/typesafe-ai/jev`), on the Upstash-provided key by default, named input variables, scoped target discovery, action timeouts, a configurable Jev confidence threshold (default 0.8), and variable-backed replay.
7 changes: 7 additions & 0 deletions packages/python-sdk/CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,13 @@ All notable changes to `upstash-box` (Python) are documented here.

## Unreleased

- Add Jev-only `confidence_threshold` to `tab.act()`, with default 0.8 and an inclusive 0–1 range.

- Support Jev browser actions through `model="jev"` (also `"typesafe-ai/jev"` and
`"vercel/typesafe-ai/jev"`), on the Upstash-provided key by default. Add
`variables`, `scope`, and `timeout` to `tab.act()`, including variable-backed
deterministic action replay.

- Fix `delete_boxes(box_ids=[])` deleting every box on the account. The API read
an empty id list as "no filter". `delete_boxes` and `delete_snapshots` now raise
`BoxError` before any request is made when the list is empty or contains a
Expand Down
5 changes: 5 additions & 0 deletions packages/python-sdk/PARITY.md
Original file line number Diff line number Diff line change
Expand Up @@ -10,8 +10,13 @@ JS `Run`/`StreamRun` → Python `Run`/`StreamRun` (+ `AsyncRun`/`AsyncStreamRun`

## Module exports

Browser act options: JS `BrowserActOptions` / `BrowserActReplayOptions` map to
Python `Tab.act(..., model=, variables=, scope=, timeout=, confidence_threshold=)` keyword arguments.
Both SDKs accept `jev`, `typesafe-ai/jev` and `vercel/typesafe-ai/jev` for Jev and preserve `%name%` action arguments.

| JS | Python |
| ---------------------- | ---------------------------- |
| `BrowserActOptions` / `BrowserActReplayOptions` | `Tab.act()` keyword arguments: `model`, `variables`, `scope`, `timeout`, `confidence_threshold` |
| `Box` / `EphemeralBox` | `Box` / `EphemeralBox` (+ `Async*`) |
| `Run` / `StreamRun` | `Run` / `StreamRun` (+ `Async*`) |
| `BoxError` | `BoxError` |
Expand Down
19 changes: 19 additions & 0 deletions packages/python-sdk/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -345,3 +345,22 @@ value.
## License

MIT


### Jev decision threshold

```python
result = await tab.act(
"Open Edit profile, fill Display name with %name%, and Save profile",
model="jev", # Also "typesafe-ai/jev" or "vercel/typesafe-ai/jev".
variables={"name": "Ada"},
confidence_threshold=0.7, # Optional; default 0.8. Also available on the sync client.
)
```

Jev runs on the Upstash-provided key by default, billed at $0.042 per million
input tokens with free output; a non-managed Box with your own Vercel AI Gateway
key uses that key instead. The threshold accepts finite numbers from 0 to 1 inclusive. Lower values accept
more uncertain action and completion decisions. Target validation and execution
limits remain enforced. This option is only supported for Jev instructions,
not action replay or other models. Check `result.success` and `result.message`.
102 changes: 102 additions & 0 deletions packages/python-sdk/tests/_async/test_box_browser.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,61 @@
BASE = f"{TEST_BASE_URL}/v2/box/box-123"


@respx.mock
async def test_jev_act_options_and_variable_replay():
box = await make_async_box(respx.mock)
route = respx.post(f"{BASE}/browser/act").mock(
return_value=httpx.Response(
200,
json={
"success": True,
"input_tokens": 100,
"output_tokens": 5,
"actions": [
{
"selector": "#email",
"description": "Email",
"method": "fill",
"arguments": ["%email%"],
}
],
},
)
)
tab = box.browser.get_tab("tab-1")
result = await tab.act(
"Fill Email with %email%",
model="vercel/typesafe-ai/jev",
variables={"email": "hello@example.com"},
scope="#login",
timeout=15000,
confidence_threshold=0.7,
)
assert last_json_body(route) == {
"instruction": "Fill Email with %email%",
"model": "vercel/typesafe-ai/jev",
"tab": "tab-1",
"variables": {"email": "hello@example.com"},
"scope": "#login",
"timeout": 15000,
"confidence_threshold": 0.7,
}
assert result.actions[0].arguments == ["%email%"]
await tab.act(result.actions[0], variables={"email": "other@example.com"})
assert "model" not in last_json_body(route)
assert last_json_body(route)["variables"] == {"email": "other@example.com"}
await box.aclose()


@pytest.mark.parametrize("timeout", [0, -1, 1.5, 180001, True])
@respx.mock
async def test_invalid_act_timeout(timeout):
box = await make_async_box(respx.mock)
with pytest.raises(BoxError, match="act timeout"):
await box.browser.get_tab("tab-1").act("Click Submit", timeout=timeout)
await box.aclose()


# ---------- tabs / page operations ----------


Expand Down Expand Up @@ -581,3 +636,50 @@ async def test_recordings_list_paginates_and_get():
assert "limit=100" in str(first)
assert "cursor=cursor-2" in str(second)
await box.aclose()


@pytest.mark.parametrize("threshold", [-0.1, 1.1, float("nan"), float("inf"), True, "0.7"])
@respx.mock
async def test_invalid_jev_confidence_threshold(threshold):
box = await make_async_box(respx.mock)
with pytest.raises(BoxError, match="confidence threshold"):
await box.browser.get_tab("tab-1").act(
"Click Submit", model="vercel/typesafe-ai/jev", confidence_threshold=threshold
)
await box.aclose()


@pytest.mark.parametrize("threshold", [0, 1])
@respx.mock
async def test_jev_threshold_boundaries(threshold):
box = await make_async_box(respx.mock)
route = respx.post(f"{BASE}/browser/act").mock(
return_value=httpx.Response(200, json={"success": True})
)
await box.browser.get_tab("tab-1").act(
"Click Submit", model="vercel/typesafe-ai/jev", confidence_threshold=threshold
)
assert last_json_body(route)["confidence_threshold"] == threshold
await box.aclose()


@pytest.mark.parametrize("model", ["jev", "typesafe-ai/jev"])
@respx.mock
async def test_jev_threshold_accepts_aliases(model):
box = await make_async_box(respx.mock)
route = respx.post(f"{BASE}/browser/act").mock(
return_value=httpx.Response(200, json={"success": True})
)
await box.browser.get_tab("tab-1").act("Click Submit", model=model, confidence_threshold=0.7)
body = last_json_body(route)
assert body["model"] == model
assert body["confidence_threshold"] == 0.7
await box.aclose()


@respx.mock
async def test_threshold_rejects_other_models():
box = await make_async_box(respx.mock)
with pytest.raises(BoxError, match="only for Jev instructions"):
await box.browser.get_tab("tab-1").act("Click Submit", confidence_threshold=0.7)
await box.aclose()
38 changes: 38 additions & 0 deletions packages/python-sdk/tests/_sync/test_sync_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,44 @@
RUN_URL = f"{BASE}/run/stream"


@respx.mock
def test_jev_act_options_and_replay():
box = make_sync_box(respx.mock)
route = respx.post(f"{BASE}/browser/act").mock(
return_value=httpx.Response(
200,
json={
"success": True,
"actions": [
{
"selector": "#email",
"description": "Email",
"method": "fill",
"arguments": ["%email%"],
}
],
},
)
)
tab = box.browser.get_tab("tab-1")
result = tab.act(
"Fill Email with %email%",
model="vercel/typesafe-ai/jev",
variables={"email": "hello@example.com"},
confidence_threshold=0.7,
scope="#login",
timeout=15000,
)
assert last_json_body(route)["confidence_threshold"] == 0.7
assert result.success
assert last_json_body(route)["timeout"] == 15000
assert last_json_body(route)["scope"] == "#login"
tab.act(result.actions[0], variables={"email": "other@example.com"})
assert "model" not in last_json_body(route)
assert last_json_body(route)["variables"] == {"email": "other@example.com"}
box.close()


def _opts():
return {"api_key": TEST_API_KEY, "base_url": TEST_BASE_URL}

Expand Down
35 changes: 33 additions & 2 deletions packages/python-sdk/upstash_box/_async/client.py
Original file line number Diff line number Diff line change
Expand Up @@ -106,6 +106,8 @@

WORKSPACE = common.WORKSPACE
_DEFAULT_TIMEOUT_MS = 600000
# Model names that select Jev for browser act.
_JEV_MODELS = frozenset({"jev", "typesafe-ai/jev", "vercel/typesafe-ai/jev"})


def _resolve_tool_call_id(parsed: Dict[str, Any]) -> Optional[str]:
Expand Down Expand Up @@ -690,25 +692,54 @@ async def act(
instruction: Union[str, BrowserObserveElement, BrowserActAction],
*,
model: Optional[str] = None,
variables: Optional[Dict[str, str]] = None,
scope: Optional[str] = None,
timeout: Optional[int] = None,
confidence_threshold: Optional[float] = None,
) -> BrowserActResult:
"""Resolve and execute one action on this tab.
"""Execute a focused instruction on this tab, possibly using several interactions.

Pass a string (LLM-resolved, metered) or a pre-resolved ``observe()``
action to replay it with no LLM call and no key (``model`` ignored).
Jev's ``confidence_threshold`` is from 0 to 1 inclusive, default 0.8.
Lower values accept more uncertain model decisions; target checks still apply.
"""
if confidence_threshold is not None:
if (
isinstance(confidence_threshold, bool)
or not isinstance(confidence_threshold, (int, float))
or not 0 <= confidence_threshold <= 1
):
raise BoxError("act confidence threshold must be a finite number from 0 to 1")
if not isinstance(instruction, str) or model not in _JEV_MODELS:
raise BoxError("act confidence threshold is supported only for Jev instructions")
if timeout is not None and (
isinstance(timeout, bool) or not isinstance(timeout, int) or not 1 <= timeout <= 180000
):
raise BoxError("act timeout must be an integer from 1 to 180000 milliseconds")
if scope is not None and not scope.strip():
raise BoxError("act scope must be a non-empty CSS selector")
if isinstance(instruction, str):
body: Dict[str, Any] = {"instruction": instruction, "tab": self.id}
if model:
body["model"] = model
if scope:
body["scope"] = scope
else:
if not instruction.selector:
raise BoxError("act(action) requires a selector; observe() did not resolve one")
body = {"action": instruction.model_dump(exclude_none=True), "tab": self.id}
if confidence_threshold is not None:
body["confidence_threshold"] = confidence_threshold
if variables is not None:
body["variables"] = variables
if timeout is not None:
body["timeout"] = timeout
resp = await self._box._request(
"POST",
f"/v2/box/{self._box.id}/browser/act",
body=body,
timeout=180000,
timeout=timeout + 5000 if timeout is not None else 185000,
)
return BrowserActResult.model_validate(resp)

Expand Down
35 changes: 33 additions & 2 deletions packages/python-sdk/upstash_box/_sync/client.py
Original file line number Diff line number Diff line change
Expand Up @@ -105,6 +105,8 @@

WORKSPACE = common.WORKSPACE
_DEFAULT_TIMEOUT_MS = 600000
# Model names that select Jev for browser act.
_JEV_MODELS = frozenset({"jev", "typesafe-ai/jev", "vercel/typesafe-ai/jev"})


def _resolve_tool_call_id(parsed: Dict[str, Any]) -> Optional[str]:
Expand Down Expand Up @@ -683,25 +685,54 @@ def act(
instruction: Union[str, BrowserObserveElement, BrowserActAction],
*,
model: Optional[str] = None,
variables: Optional[Dict[str, str]] = None,
scope: Optional[str] = None,
timeout: Optional[int] = None,
confidence_threshold: Optional[float] = None,
) -> BrowserActResult:
"""Resolve and execute one action on this tab.
"""Execute a focused instruction on this tab, possibly using several interactions.

Pass a string (LLM-resolved, metered) or a pre-resolved ``observe()``
action to replay it with no LLM call and no key (``model`` ignored).
Jev's ``confidence_threshold`` is from 0 to 1 inclusive, default 0.8.
Lower values accept more uncertain model decisions; target checks still apply.
"""
if confidence_threshold is not None:
if (
isinstance(confidence_threshold, bool)
or not isinstance(confidence_threshold, (int, float))
or not 0 <= confidence_threshold <= 1
):
raise BoxError("act confidence threshold must be a finite number from 0 to 1")
if not isinstance(instruction, str) or model not in _JEV_MODELS:
raise BoxError("act confidence threshold is supported only for Jev instructions")
if timeout is not None and (
isinstance(timeout, bool) or not isinstance(timeout, int) or not 1 <= timeout <= 180000
):
raise BoxError("act timeout must be an integer from 1 to 180000 milliseconds")
if scope is not None and not scope.strip():
raise BoxError("act scope must be a non-empty CSS selector")
if isinstance(instruction, str):
body: Dict[str, Any] = {"instruction": instruction, "tab": self.id}
if model:
body["model"] = model
if scope:
body["scope"] = scope
else:
if not instruction.selector:
raise BoxError("act(action) requires a selector; observe() did not resolve one")
body = {"action": instruction.model_dump(exclude_none=True), "tab": self.id}
if confidence_threshold is not None:
body["confidence_threshold"] = confidence_threshold
if variables is not None:
body["variables"] = variables
if timeout is not None:
body["timeout"] = timeout
resp = self._box._request(
"POST",
f"/v2/box/{self._box.id}/browser/act",
body=body,
timeout=180000,
timeout=timeout + 5000 if timeout is not None else 185000,
)
return BrowserActResult.model_validate(resp)

Expand Down
Loading
Loading