Skip to content
11 changes: 5 additions & 6 deletions pageindex/agent_tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -1596,7 +1596,7 @@ def build_agent_tools(client, include_management: bool = False,

_DISCOVERY = """\
DOCUMENT DISCOVERY:
- browse_documents() — DEFAULT discovery tool, first choice for any document-related question. It lists your documents newest first with names and descriptions; match them against the user's intent, and page through with `offset: next_offset` while has_more is true."""
- browse_documents() — DEFAULT discovery tool, first choice for any document-related question. The bare call returns your documents newest first with names and descriptions; match them against the user's intent."""

_DECISION = """\
DECISION:
Expand All @@ -1612,9 +1612,9 @@ def build_agent_tools(client, include_management: bool = False,
PERSISTENCE (before concluding the target document is not in the library):
This protocol applies both when results are empty AND when results are returned but none match the user's intent. Do NOT give up after a single discovery attempt. Follow these steps in order:
1. browse_documents() and compare every returned name/description against the user's intent
2. Page through the ENTIRE library with `limit: 50` and `offset: next_offset` until has_more is false — MANDATORY, must be completed before concluding "not found"
3. Re-scan for loose matches: synonyms, abbreviations, and partial titles in names/descriptions can identify the target
Only after ALL three steps have been tried may you conclude the document is not in the library. Do NOT fall back to general knowledge — if the user's question references their own documents, exhaust every discovery path first."""
2. Rephrase the query with synonyms or alternative terms and browse again
3. Page through the ENTIRE library with `limit: 50` and `offset: next_offset` until has_more is false — MANDATORY, must be completed before concluding "not found"
Only after ALL steps have been tried may you conclude the document is not in the library. Do NOT fall back to general knowledge — if the user's question references their own documents, exhaust every discovery path first."""

AGENT_INSTRUCTIONS = "\n\n".join([
_INSTRUCTIONS_HEADER,
Expand Down Expand Up @@ -1651,8 +1651,7 @@ def build_agent_tools(client, include_management: bool = False,
- Cite only statements supported by tool outputs: <cite doc="{docName}" page="{pageNumber}"/> or <cite doc="{docName}" page="{pageNumber}" block="{blockId}"/>. Place immediately after the claim.
- When page content includes block_id values, citations MUST be block-level: copy the exact block_id of the supporting block. Page-only cites are allowed ONLY when the tool output carries no block_id (legacy documents, structure outlines). NEVER invent or alter block_id values.
- For a claim drawn from multiple blocks on one page, add one tag per supporting block (at most 3); beyond that, cite the single strongest block.
- Each tag must reference a SINGLE page integer. For multi-page citations, use separate tags.
- Close every answer with a "Sources" section: one plain-text line per cited block, formatted `- <document name>, page <page> (block <block_id>)`. Keep it even when the inline tags are present — a client that strips unknown HTML tags would otherwise leave the answer with no visible citations at all.""",
- Each tag must reference a SINGLE page integer. For multi-page citations, use separate tags.""",
"footnote": """\
GROUNDING
- Answer only from the user's PageIndex documents. Call get_page_content() and state only what was actually read there.
Expand Down
13 changes: 13 additions & 0 deletions pageindex/client.py
Original file line number Diff line number Diff line change
Expand Up @@ -1653,6 +1653,19 @@ def get_document(self, doc_id: str) -> dict[str, Any]:
"""
return self._api.get_document(doc_id=doc_id)

def get_document_id(self, name: str) -> str:
"""
Look up a document's ID by its name. Useful for resolving
citation doc names (from ``<cite doc="…">``) to IDs.

Raises PageIndexAPIError if no document with that name exists.
"""
result = self._api.list_documents(limit=1, name=name)
docs = result.get("documents", [])
if docs:
return docs[0]["id"]
raise PageIndexAPIError(f"No document named {name!r} found.")

def delete_document(self, doc_id: str) -> dict[str, Any]:
"""
Delete a PageIndex document and all its associated data.
Expand Down
4 changes: 3 additions & 1 deletion pageindex/cloud_api.py
Original file line number Diff line number Diff line change
Expand Up @@ -369,7 +369,7 @@ def delete_document(self, doc_id: str) -> Dict[str, Any]:
status_code=response.status_code)
return response.json() if response.content else {}

def list_documents(self, limit: int = 50, offset: int = 0, folder_id: Optional[str] = None) -> Dict[str, Any]:
def list_documents(self, limit: int = 50, offset: int = 0, folder_id: Optional[str] = None, name: Optional[str] = None) -> Dict[str, Any]:
"""
List all documents for the authenticated user with pagination.

Expand All @@ -394,6 +394,8 @@ def list_documents(self, limit: int = 50, offset: int = 0, folder_id: Optional[s
params: Dict[str, Any] = {"limit": limit, "offset": offset}
if folder_id is not None:
params["folder_id"] = folder_id
if name is not None:
params["name"] = name

response = requests.get(
f"{self.BASE_URL}/docs/",
Expand Down
16 changes: 14 additions & 2 deletions pageindex/flash/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,14 +40,26 @@ Writes the tree to `results/<name>_structure.json`.
"node_id": str, # 4-digit, zero-padded
"start_index": int, # 1-based, inclusive
"end_index": int,
"summary": str,
"summary": str, # summary=True only
"key_items": [str], # optimize only: titles of subsections merged away
"nodes": [...],
"nodes": [...], # entries with children only
}
],
"toc_source": str, # "detected" | "bookmarks" | "hybrid" | "pages" | "unreadable"
}
```

`toc_source` says where the structure came from: `"detected"` from the layout,
`"bookmarks"` from the embedded outline, `"hybrid"` when bookmarks frame the
detected sections. `"pages"` means the layout yielded no hierarchy, so every
page became one node titled `Page N`; past `FLAT_TREE_MAX_NODES` (10) pages that
flat tree comes back without summaries or optimization, and the local client and
CLI refuse it. `"unreadable"` means no page carries text and `structure` is
empty.

Every page is in some node: a hierarchy that starts after page 1 is preceded by
a `Preface` node covering the pages before it, as in standard mode.

## Benchmark

Nine PDFs, each run end to end with tree optimization: PDF parse, layout
Expand Down
57 changes: 55 additions & 2 deletions pageindex/flash/api.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
"""Public API for PageIndex Flash. The only supported entry point is :func:`page_index_flash`. Everything else in this package is internal pipeline machinery."""
"""Public API for PageIndex Flash: :func:`page_index_flash` builds the tree, :func:`flash_rejection_reason` is the refusal policy the local client and CLI share. Everything else in this package is internal pipeline machinery."""

from __future__ import annotations

Expand All @@ -10,6 +10,9 @@

from .main import extract_toc

# Largest page-node fallback the managed pipelines accept as an index.
FLAT_TREE_MAX_NODES = 10


def _is_pdfium_password_error(exc: Exception) -> bool:
msg = str(exc).lower()
Expand Down Expand Up @@ -95,11 +98,50 @@ def _optimize(structure, page_texts, do_expand, model):
"before": outcome["before"], "after": outcome["after"]}


def _page_nodes(page_texts: list[str]) -> list[dict]:
"""One node per page, so every page is reachable."""
from ..utils import write_node_id
if not any(text.strip() for text in page_texts):
return []
nodes = [{"title": f"Page {index}", "node_id": "", "start_index": index,
"end_index": index} for index in range(1, len(page_texts) + 1)]
write_node_id(nodes)
return nodes


def _add_preface(structure: list[dict]) -> None:
"""The pages before a hierarchy that starts late become a Preface node, as in standard mode."""
from ..utils import write_node_id
structure.insert(0, {"title": "Preface", "start_index": 1,
"end_index": structure[0]["start_index"] - 1})
write_node_id(structure)


def flash_rejection_reason(result: dict, standard_hint: str = "mode='standard'") -> str | None:
"""Why a managed pipeline should refuse this flash result, or None to accept it.

The local client and the CLI share this policy so they refuse the same
documents; ``standard_hint`` is how each spells the standard-mode switch.
"""
structure = result.get("structure") or []
if result.get("toc_source") == "unreadable":
return ("PageIndex Flash found no text layer in this PDF (scanned or "
"image-only); run OCR before indexing it.")
if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES:
return (f"PageIndex Flash found no layout structure in this document "
f"({len(structure)} pages); try {standard_hint}, which builds "
"the structure with the model.")
if not structure:
return ("PageIndex Flash could not extract a structure from this PDF; "
f"try {standard_hint}, which builds the structure with the model.")
return None


def page_index_flash(pdf, summary=True, summary_model=None,
optimize: str | bool | None = None, optimize_expand=None,
optimize_model=None, summary_concurrency=None,
use_embedded_toc=True) -> dict:
"""Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored; adds a ``toc_source`` key to the result. On by default; pass False for the pure detected structure. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """
"""Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """
if optimize_expand is not None:
import warnings
warnings.warn(
Expand All @@ -118,6 +160,17 @@ def page_index_flash(pdf, summary=True, summary_model=None,
f"optimize must be 'full', 'merge', or False, got {optimize!r}")
result = extract_toc(_validate_pdf(pdf), use_embedded_toc=use_embedded_toc)
structure = result.get("structure", [])
if not structure:
# the layout yields no hierarchy; the pages themselves are the tree
structure = _page_nodes(result.get("page_texts") or [])
result["structure"] = structure
result["toc_source"] = "pages" if structure else "unreadable"
elif structure[0]["start_index"] > 1:
_add_preface(structure)
if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES:
# the managed pipelines refuse a flat tree this size; skip the model passes
result.pop("page_texts", None)
return result
if optimize and structure:
# bookmark-only extractions carry no page_texts and scanned ones
# only empty strings; expand needs text
Expand Down
8 changes: 2 additions & 6 deletions pageindex/flash/heading_detection/page_scan.py
Original file line number Diff line number Diff line change
Expand Up @@ -295,7 +295,6 @@ def __init__(self, doc, labeled):

def filter_page_candidates(doc_collector: DocCandidateCollector, page, page_candidates: list[HeadingCandidate]) -> None:
"""Per-page candidate filter for noisy pages, title overlap, page headers, and numbering continuity."""
from ..outline_assembly import is_script_compatible
from ..model import intervals_overlap, is_caps_heavy
from ..stats import column_index_of

Expand Down Expand Up @@ -350,9 +349,6 @@ def _ih_key(heading_candidate: HeadingCandidate):
active_candidate = page_candidates[index]
next_item = page_candidates[index + 1] if index + 1 < count else None

if is_script_compatible(doc_collector.previous_slot.secondary_slot.tertiary_slot, active_candidate):
continue

if not doc_collector.auxiliary_slot and active_candidate.type != 11:
bottom = active_candidate.group_slot.top_edge()
if (title is not None and bottom > title.top_edge()
Expand Down Expand Up @@ -428,7 +424,7 @@ def build_doc_heading_candidates(doc, labeled: Optional[list] = None) -> list[He

def find_section_openers(doc, start_page_idx: int) -> list:
"""Find the first valid heading on each page, then clique-filter the result."""
from ..outline_assembly import is_script_compatible, has_conflict_in_context, OutlineContext, OutlineNode
from ..outline_assembly import has_conflict_in_context, OutlineContext, OutlineNode

item_list: list[HeadingCandidate] = []
index = start_page_idx
Expand All @@ -450,7 +446,7 @@ def find_section_openers(doc, start_page_idx: int) -> list:
break
if block.line_count() > 2:
break
if current_candidate is not None and not is_script_compatible(doc.secondary_slot.tertiary_slot, current_candidate):
if current_candidate is not None:
item_list.append(current_candidate)
index += 1

Expand Down
Loading
Loading