Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 22 additions & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -21,12 +21,13 @@ jobs:
working-directory: backend
- run: uv run python -c "import pixelbot, pixeltable; assert pixelbot.__version__ == '3.0.0'; assert pixeltable.__version__ == '0.7.7'"
working-directory: backend
- run: uv run pytest -q
working-directory: backend
- if: matrix.python-version == '3.11'
run: |
uv run ruff check pixelbot tests
uv run ruff format --check pixelbot tests
uv run mypy pixelbot
uv run pytest -q
working-directory: backend

frontend:
Expand Down Expand Up @@ -56,8 +57,28 @@ jobs:
python-version: "3.11"
- run: uv sync --locked --group dev
working-directory: backend
- name: Verify fresh dependency resolution
working-directory: backend
run: |
uv pip compile pyproject.toml --python-version 3.11 --output-file "$RUNNER_TEMP/fresh-requirements.txt"
grep -Fx "pixeltable==0.7.7" "$RUNNER_TEMP/fresh-requirements.txt"
- run: uv build
working-directory: backend
- name: Verify source distribution contents
working-directory: backend
run: |
uv run python - <<'PY'
import tarfile
from pathlib import Path, PurePosixPath

archive = next(Path("dist").glob("pixelbot-*.tar.gz"))
with tarfile.open(archive) as package:
files = [PurePosixPath(member.name) for member in package.getmembers() if member.isfile()]
forbidden = {"uv-bin", "uv-cache", "dist", "static", ".venv"}
leaked = [path for path in files if len(path.parts) > 1 and path.parts[1] in forbidden]
assert not leaked, leaked[:10]
assert len(files) < 200, f"Unexpectedly large sdist: {len(files)} files"
PY
- run: |
uv venv wheel-test
uv pip install --python wheel-test/bin/python dist/pixelbot-3.0.0-py3-none-any.whl --no-deps
Expand Down
2 changes: 1 addition & 1 deletion AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ uv run --env-file .env pxt schema update pixelbot/app.py pixelbot_v3 -f
uv run --env-file .env pxt service update pixelbot/app.py pixelbot_v3 app --port 8000 -f
```

Debug with `pxt service list`, `pxt service logs pixelbot_v3/app`, `pxt errors TABLE`, and `pxt recompute TABLE COLUMN --errors-only -f`. Errors-only recomputation takes exactly one column.
Debug with `pxt service list`, the local file at `$PIXELTABLE_HOME/logs/services/pixelbot_v3/app.log`, `pxt errors TABLE`, and `pxt recompute TABLE COLUMN --errors-only -f`. The local `pxt service logs` command reports that path rather than streaming the file. Errors-only recomputation takes exactly one column.

## Required checks

Expand Down
4 changes: 3 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -47,14 +47,16 @@ cd backend
export PIXELTABLE_HOME="$PWD/.pixeltable-v3"

uv run pxt service list
uv run pxt service logs pixelbot_v3/app --since 10m --tail 50
tail -50 "$PIXELTABLE_HOME/logs/services/pixelbot_v3/app.log"
uv run pxt errors pixelbot_v3/collection
uv run pxt recompute pixelbot_v3/collection summary --errors-only -f
uv run pxt service stop pixelbot_v3/app
```

`--errors-only` accepts exactly one computed column. A computed expression cannot be migrated in place: rename the column, or drop and re-add it in separate schema changes. `--allow-destructive` does not make that migration supported.

For a local service, `pxt service logs pixelbot_v3/app` reports the log file path but does not stream its contents. Read the file under `PIXELTABLE_HOME` as shown above.

The Database page is an inspector for catalog rows, schemas, lineage, history, samples, joins, and computation errors. Schema and index changes are made in `pixelbot/schema.py` and applied with `pxt schema update`. CSV uploads remain editable only through their registry UUIDs in `pixelbot_scratch`.

## Safety boundaries
Expand Down
6 changes: 1 addition & 5 deletions backend/pixelbot/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,20 +38,16 @@
UPLOAD_FOLDER = "data"
MAX_UPLOAD_SIZE_MB = 100
ALLOWED_EXTENSIONS = {
# Documents (native + Office via MarkdownIT)
# Documents supported by Pixeltable 0.7.7's DocumentType
"pdf",
"txt",
"md",
"html",
"xml",
"doc",
"docx",
"ppt",
"pptx",
"xls",
"xlsx",
"csv",
"rtf",
# Images
"jpg",
"jpeg",
Expand Down
8 changes: 4 additions & 4 deletions backend/pixelbot/functions.py
Original file line number Diff line number Diff line change
Expand Up @@ -226,13 +226,13 @@ def extract_document_text(doc: pxt.Document) -> Optional[str]:
pages = [page.extract_text() or "" for page in pdf.pages]
text = "\n".join(pages)

elif ext in (".docx", ".doc"):
elif ext == ".docx":
import docx as docx_lib

doc_obj = docx_lib.Document(str(path))
text = "\n".join(p.text for p in doc_obj.paragraphs if p.text.strip())

elif ext in (".pptx", ".ppt"):
elif ext == ".pptx":
from pptx import Presentation

prs = Presentation(str(path))
Expand All @@ -243,7 +243,7 @@ def extract_document_text(doc: pxt.Document) -> Optional[str]:
slide_texts.append(shape.text.strip())
text = "\n".join(slide_texts)

elif ext in (".xlsx", ".xls"):
elif ext == ".xlsx":
import openpyxl

wb = openpyxl.load_workbook(str(path), read_only=True, data_only=True)
Expand All @@ -269,7 +269,7 @@ def extract_document_text(doc: pxt.Document) -> Optional[str]:
break
text = "\n".join(csv_rows)

elif ext in (".txt", ".md", ".html", ".xml", ".rtf"):
elif ext in (".txt", ".md", ".html", ".xml"):
text = path.read_text(errors="ignore")

else:
Expand Down
2 changes: 1 addition & 1 deletion backend/pixelbot/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -452,7 +452,7 @@ class NotificationRow(BaseModel):
message: str
status: str
response_code: int
timestamp: datetime = Field(default_factory=datetime.utcnow)
timestamp: datetime = Field(default_factory=datetime.now)
user_id: str = config.DEFAULT_USER_ID


Expand Down
62 changes: 32 additions & 30 deletions backend/pixelbot/routers/database.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,12 +2,13 @@

import logging
import re
from collections.abc import Mapping
from datetime import datetime
from typing import Literal, cast
from typing import Any, Literal, cast

import pixeltable as pxt
from fastapi import APIRouter, HTTPException
from pydantic import BaseModel
from fastapi import APIRouter, HTTPException, Query
from pydantic import BaseModel, Field

from pixelbot import config
from pixelbot.catalog_access import require_allowed_table
Expand Down Expand Up @@ -60,10 +61,21 @@ def _column_info(tbl) -> list[dict]:
return columns


def _index_info(name: str, metadata: Mapping[str, Any]) -> dict:
"""Normalize Pixeltable index metadata for the pipeline response."""
parameters = metadata.get("parameters") or {}
return {
"name": name,
"columns": metadata.get("columns", []),
"type": metadata.get("index_type", "unknown"),
"embedding": str(parameters.get("embedding", ""))[:120],
}


@router.get("/tables")
@pxt_retry()
def list_all_tables():
"""List all tables and views in the agents namespace with schema info."""
"""List all tables and views in the application namespace with schema info."""
table_paths = pxt.list_tables(NAMESPACE, recursive=True)

tables = []
Expand Down Expand Up @@ -92,7 +104,7 @@ def list_all_tables():
"base_table": None,
"columns": [],
"row_count": 0,
"error": str(e),
"error": "Unable to inspect table",
}
)

Expand Down Expand Up @@ -135,11 +147,11 @@ def get_table_rows(path: str, limit: int = 50, offset: int = 0):

class SampleRequest(BaseModel):
path: str
n: int | None = None
fraction: float | None = None
n: int | None = Field(default=None, ge=1)
fraction: float | None = Field(default=None, gt=0.0, le=1.0)
stratify_by: str | None = None
seed: int | None = 42
limit: int = 100
limit: int = Field(default=100, ge=1, le=100)


@router.post("/sample")
Expand Down Expand Up @@ -178,7 +190,7 @@ def sample_table(body: SampleRequest):
)
sample_kwargs["stratify_by"] = getattr(tbl, body.stratify_by)

raw_rows = query.sample(**sample_kwargs).collect()
raw_rows = query.sample(**sample_kwargs).limit(body.limit).collect()

col_names = tbl.columns()
rows = []
Expand All @@ -203,7 +215,7 @@ def sample_table(body: SampleRequest):

@router.get("/timeline")
@pxt_retry()
def get_timeline(limit: int = 100):
def get_timeline(limit: int = Query(default=100, ge=1, le=500)):
"""Unified chronological feed across all timestamped tables."""
events: list[dict] = []

Expand Down Expand Up @@ -279,8 +291,8 @@ class JoinRequest(BaseModel):
right_table: str
left_column: str
right_column: str
join_type: str = "inner" # inner, left, cross
limit: int = 50
join_type: Literal["inner", "left", "cross"] = "inner"
limit: int = Field(default=50, ge=1, le=100)


@router.post("/join")
Expand All @@ -306,8 +318,6 @@ def join_tables(body: JoinRequest):
if body.right_column not in right_cols:
raise HTTPException(status_code=400, detail=f"Column '{body.right_column}' not in {body.right_table}")

if body.join_type not in ("inner", "left", "cross"):
raise HTTPException(status_code=400, detail=f"Unsupported join type: {body.join_type}")
left_col_ref = getattr(left, body.left_column)
right_col_ref = getattr(right, body.right_column)

Expand Down Expand Up @@ -432,14 +442,14 @@ def _parse_deps(computed_with: str | None, all_cols: set[str]) -> list[str]:

def _detect_iterator(columns: list[dict]) -> str | None:
"""Detect the iterator type used to create a view from its column shapes."""
own_cols = {c["name"] for c in columns if c.get("defined_in_self")}
if {"frame_idx", "pos_frame", "frame"} & own_cols:
iterator_cols = {c["name"] for c in columns if c.get("is_iterator_col")}
if {"pos", "frame_attrs", "frame"} <= iterator_cols:
return "FrameIterator"
if {"audio_chunk"} & own_cols and {"start_time_sec", "end_time_sec"} & own_cols:
if {"audio_segment", "segment_start", "segment_end"} <= iterator_cols:
return "AudioSplitter"
if {"heading", "page", "title"} & own_cols and "pos" in own_cols:
if {"heading", "page", "title"} & iterator_cols and "pos" in iterator_cols:
return "DocumentSplitter"
if "text" in own_cols and "pos" in own_cols:
if {"text", "pos"} <= iterator_cols:
return "StringSplitter"
return None

Expand Down Expand Up @@ -501,6 +511,7 @@ def get_pipeline():
"computed_with": cw_str,
"defined_in": defined_in,
"defined_in_self": defined_in == short_name,
"is_iterator_col": bool(info.get("is_iterator_col")),
"func_name": func_name,
"func_type": func_type,
}
Expand Down Expand Up @@ -532,16 +543,7 @@ def get_pipeline():

# Indices
raw_indexes = md.get("indexes", {})
indexes = []
for idx_name, idx_info in raw_indexes.items():
indexes.append(
{
"name": idx_name,
"columns": idx_info.get("columns", []),
"type": idx_info.get("index_type", "unknown"),
"embedding": str(idx_info.get("parameters", {}).get("embedding", ""))[:120],
}
)
indexes = [_index_info(idx_name, idx_metadata) for idx_name, idx_metadata in raw_indexes.items()]

# Version history (last 10)
try:
Expand Down Expand Up @@ -629,7 +631,7 @@ def get_pipeline():
"versions": [],
"computed_count": 0,
"insertable_count": 0,
"error": str(e),
"error": "Unable to inspect table",
}
)

Expand Down
12 changes: 6 additions & 6 deletions backend/pixelbot/routers/export.py
Original file line number Diff line number Diff line change
Expand Up @@ -95,7 +95,7 @@ def list_exportable_tables():
@pxt_retry()
def export_json(
table_path: str,
limit: int = Query(default=1000, le=1000),
limit: int = Query(default=1000, ge=1, le=1000),
columns: str | None = Query(default=None, description="Comma-separated column names"),
):
"""Export a table as a downloadable JSON file."""
Expand All @@ -119,7 +119,7 @@ def export_json(
@pxt_retry()
def export_csv(
table_path: str,
limit: int = Query(default=1000, le=1000),
limit: int = Query(default=1000, ge=1, le=1000),
columns: str | None = Query(default=None, description="Comma-separated column names"),
):
"""Export a table as a downloadable CSV file."""
Expand Down Expand Up @@ -155,7 +155,7 @@ def export_csv(
@pxt_retry()
def export_parquet(
table_path: str,
limit: int = Query(default=1000, le=1000),
limit: int = Query(default=1000, ge=1, le=1000),
columns: str | None = Query(default=None, description="Comma-separated column names"),
):
"""Export a table as a downloadable Parquet file."""
Expand Down Expand Up @@ -195,7 +195,7 @@ def export_parquet(
def export_json_column(
table_path: str,
column: str = Query(..., description="Column to serialize"),
limit: int = Query(default=100, le=500),
limit: int = Query(default=100, ge=1, le=500),
):
"""Serialize a complex column to JSON strings using Pixeltable's json.dumps() UDF."""
try:
Expand Down Expand Up @@ -232,7 +232,7 @@ def export_json_column(
@pxt_retry()
def preview_table(
table_path: str,
limit: int = Query(default=5, le=50),
limit: int = Query(default=5, ge=1, le=50),
columns: str | None = Query(default=None),
):
"""Return a small preview of a table for the export UI."""
Expand All @@ -249,7 +249,7 @@ def preview_table(
def export_native(
table_path: str,
format: str = Query(default="csv", pattern="^(csv|json)$"),
limit: int = Query(default=1000, le=1000),
limit: int = Query(default=1000, ge=1, le=1000),
):
"""Export using Pixeltable 0.7.7 native export APIs (demo endpoint)."""
try:
Expand Down
Loading
Loading